vrex-flow-engine 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1743 @@
1
+ """Minimal Google Flow SDK wrapper.
2
+
3
+ Ported from flowkit (`/tmp/flowkit-ref/agent/services/flow_client.py`).
4
+ Trimmed to what Run 4 ships: `create_project` (TRPC) + `gen_image` (api_request
5
+ with IMAGE_GENERATION captcha). Video / upload / upscale / check_async land in
6
+ later runs.
7
+
8
+ The wrapper intentionally preserves `raw` on every return so callers (and the
9
+ request-worker that persists it to the DB) can inspect Flow's error payload
10
+ when the user's paygate tier or model name drifts.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import logging
15
+ import re
16
+ import time
17
+ import uuid
18
+ from typing import Any, Optional
19
+
20
+ from flow_engine.bridge.flow_client import FlowClient
21
+
22
+ logger = logging.getLogger(__name__)
23
+
24
+ # Endpoints -----------------------------------------------------------------
25
+
26
+ FLOW_API_BASE = "https://aisandbox-pa.googleapis.com"
27
+ TRPC_CREATE_PROJECT = "https://labs.google/fx/api/trpc/project.createProject"
28
+ TRPC_SEARCH_PROJECTS = "https://labs.google/fx/api/trpc/project.searchUserProjects"
29
+ # Returns the full modelConfig (image/video families + keys + per-tier credit
30
+ # availability), userData (serviceTier/paygateTier/credits), and the preset
31
+ # audio voice list — used to build the model catalog dynamically.
32
+ TRPC_PROJECT_INITIAL_DATA = "https://labs.google/fx/api/trpc/flow.projectInitialData"
33
+ VIDEO_I2V_URL = f"{FLOW_API_BASE}/v1/video:batchAsyncGenerateVideoStartImage"
34
+ # Reference-to-video endpoint: takes referenceImages[] (multi-ref, asset-typed)
35
+ # instead of a single startImage. See gen_video_r2v() for the body assembly.
36
+ VIDEO_R2V_URL = f"{FLOW_API_BASE}/v1/video:batchAsyncGenerateVideoReferenceImages"
37
+ # Text-to-video: same body as r2v minus referenceImages/referenceAudio.
38
+ VIDEO_T2V_URL = f"{FLOW_API_BASE}/v1/video:batchAsyncGenerateVideoText"
39
+ # Extend (continue a Veo clip) and edit (video-to-video) — both take a
40
+ # `videoInput.mediaId` pointing at a source Veo video and return the workflow
41
+ # schema (poll via the new primaryMediaId). Verified from a live labs.google
42
+ # capture (2026-05). NOTE: only Veo-generated videos can be extended.
43
+ VIDEO_EXTEND_URL = f"{FLOW_API_BASE}/v1/video:batchAsyncGenerateVideoExtendVideo"
44
+ VIDEO_EDIT_URL = f"{FLOW_API_BASE}/v1/video:batchAsyncGenerateVideoEditVideo"
45
+ VIDEO_POLL_URL = f"{FLOW_API_BASE}/v1/video:batchCheckAsyncVideoGenerationStatus"
46
+ UPLOAD_IMAGE_URL = f"{FLOW_API_BASE}/v1/flow/uploadImage"
47
+ # Video upload is a resumable, cookie-authed flow against labs.google (NOT
48
+ # aisandbox-pa): a zero-body POST ?action=start returns a session URL, then a
49
+ # binary PUT ?action=upload streams the bytes. Both run inside the extension's
50
+ # browser session (see background.js handleUploadVideo). Afterwards we register
51
+ # the clip's trim window via the TRPC videoFx.updateVideoOffset mutation so Flow
52
+ # treats the upload as a usable workflow media — mirroring the web client.
53
+ UPLOAD_VIDEO_START_URL = "https://labs.google/fx/api/upload-video?action=start"
54
+ UPLOAD_VIDEO_UPLOAD_URL = "https://labs.google/fx/api/upload-video?action=upload"
55
+ TRPC_UPDATE_VIDEO_OFFSET = "https://labs.google/fx/api/trpc/videoFx.updateVideoOffset"
56
+
57
+
58
+ # Veo 3.1 reference-to-video (r2v). Verified against a live labs.google web
59
+ # request (2026-05): Flow web dispatches `veo_3_1_r2v_lite_low_priority` to
60
+ # `batchAsyncGenerateVideoReferenceImages` and gets a workflow-mode response
61
+ # (poll via /v1/media/<primaryMediaId>). Duration is fixed by the model
62
+ # (~8s); it is not a request field.
63
+ VEO_R2V_MODEL_KEY = "veo_3_1_r2v_lite_low_priority"
64
+
65
+ R2V_VALID_ASPECTS: set[str] = {
66
+ "VIDEO_ASPECT_RATIO_PORTRAIT",
67
+ "VIDEO_ASPECT_RATIO_LANDSCAPE",
68
+ }
69
+
70
+
71
+ def _media_get_url(media_id: str) -> str:
72
+ """Endpoint that returns inline encoded video bytes for a workflow's
73
+ primary media. Used to poll Low Priority (workflow-schema) submissions —
74
+ they have no operation name and don't appear in ``batchCheckAsync``."""
75
+ return f"{FLOW_API_BASE}/v1/media/{media_id}?clientContext.tool=PINHOLE"
76
+
77
+ # Image model keys, indexed by the user-facing nickname used in
78
+ # flowkit's models.json. Pro is Flow's premium / higher-quality image
79
+ # model; "Banana 2" (NARWHAL) is the lighter / faster option. The
80
+ # frontend Settings panel lets the user pick which one drives gen_image
81
+ # + edit_image at request time. Update when Google rotates model names.
82
+ IMAGE_MODELS: dict[str, str] = {
83
+ "NANO_BANANA_PRO": "GEM_PIX_2",
84
+ "NANO_BANANA_2": "NARWHAL",
85
+ }
86
+ DEFAULT_IMAGE_MODEL_KEY = "NANO_BANANA_PRO"
87
+
88
+
89
+ def resolve_image_model(key: Optional[str]) -> str:
90
+ """Map a nickname (`NANO_BANANA_PRO` / `NANO_BANANA_2`) to the actual
91
+ Flow model identifier. Falls back to the Pro default for unknown /
92
+ missing keys so a stale frontend can't break dispatch."""
93
+ if isinstance(key, str) and key in IMAGE_MODELS:
94
+ return IMAGE_MODELS[key]
95
+ return IMAGE_MODELS[DEFAULT_IMAGE_MODEL_KEY]
96
+
97
+ # Video model keys nested by [tier][quality][aspect]. All values verified
98
+ # against real Flow web request bodies (curl exports from labs.google's
99
+ # Network tab) — do NOT speculate suffixes here, only use observed keys.
100
+ #
101
+ # `quality` is "fast" (default), "lite", "quality", or — Ultra only —
102
+ # "lite_relaxed" / "fast_relaxed" (0-credit low-priority queue).
103
+ # - Lite (`veo_3_1_i2v_lite`) is shared by Tier 1 and Tier 2; verified
104
+ # from PRO PLAN and ULTRA PLAN curls (see video_model.md and
105
+ # video_model_ultra.md). Multi-aspect — same key for both 16:9 and
106
+ # 9:16; the model adapts via the aspectRatio field.
107
+ # - Quality (`veo_3_1_i2v_s` / `veo_3_1_i2v_s_portrait`) is also shared
108
+ # across both tiers; the difference is the `userPaygateTier` in
109
+ # clientContext (rate limits / queue priority), not the model key.
110
+ # - Tier 2 Fast naming pattern: Tier 1 Fast key + `_ultra` suffix
111
+ # (e.g. `veo_3_1_i2v_s_fast` → `veo_3_1_i2v_s_fast_ultra`,
112
+ # `veo_3_1_i2v_s_fast_portrait` → `veo_3_1_i2v_s_fast_portrait_ultra`).
113
+ # - Tier 2 "low priority" 0-credit models (Ultra-only fallback when the
114
+ # user wants to keep their daily credit budget): Lite uses the
115
+ # `_low_priority` suffix (`veo_3_1_i2v_lite_low_priority`); Fast uses
116
+ # the `_relaxed` suffix on the ultra family (`veo_3_1_i2v_s_fast_ultra_relaxed`).
117
+ # Verified from ULTRA PLAN curls. PORTRAIT keys for these are not yet
118
+ # observed — we reuse the LANDSCAPE key for both aspects (Lite is
119
+ # genuinely multi-aspect; Fast Relaxed portrait will need a real curl
120
+ # to confirm, but Flow's portrait variants typically follow the
121
+ # `_portrait` suffix convention if separate keys are required).
122
+ VIDEO_MODEL_KEYS: dict[str, dict[str, dict[str, str]]] = {
123
+ # Tier 1 (Pro) — three quality levels, all verified from real PRO
124
+ # PLAN curls (see video_model.md). Lite shares `veo_3_1_i2v_lite`
125
+ # with Tier 2; Quality shares `veo_3_1_i2v_s` with Tier 2 — paygate
126
+ # tier in clientContext drives any per-tier difference. No 0-credit
127
+ # low-priority option here — that's a Tier 2 (Ultra) perk.
128
+ "PAYGATE_TIER_ONE": {
129
+ "lite": {
130
+ "VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_lite",
131
+ "VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_lite",
132
+ },
133
+ "fast": {
134
+ "VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_s_fast",
135
+ "VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_s_fast_portrait",
136
+ },
137
+ "quality": {
138
+ "VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_s",
139
+ "VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_s_portrait",
140
+ },
141
+ },
142
+ # Tier 2 (Ultra) — five quality levels:
143
+ # - lite: `veo_3_1_i2v_lite` (5 credits, multi-aspect)
144
+ # - fast: `_fast_ultra` family (10 credits, default, balanced)
145
+ # - quality: `veo_3_1_i2v_s*` family (highest fidelity, slowest)
146
+ # - lite_relaxed: `veo_3_1_i2v_lite_low_priority` (0 credits,
147
+ # low-priority queue, Ultra-only)
148
+ # - fast_relaxed: `veo_3_1_i2v_s_fast_ultra_relaxed` (0 credits,
149
+ # low-priority queue, Ultra-only).
150
+ # PORTRAIT keys for the `_relaxed` family are not yet verified
151
+ # from a real curl; we reuse the LANDSCAPE key as a best-effort
152
+ # fallback. If Flow rejects portrait dispatches, capture a portrait
153
+ # curl and add the proper key here.
154
+ "PAYGATE_TIER_TWO": {
155
+ "lite": {
156
+ "VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_lite",
157
+ "VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_lite",
158
+ },
159
+ "fast": {
160
+ "VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_s_fast_ultra",
161
+ "VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_s_fast_portrait_ultra",
162
+ },
163
+ "quality": {
164
+ "VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_s",
165
+ "VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_s_portrait",
166
+ },
167
+ "lite_relaxed": {
168
+ "VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_lite_low_priority",
169
+ "VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_lite_low_priority",
170
+ },
171
+ "fast_relaxed": {
172
+ "VIDEO_ASPECT_RATIO_LANDSCAPE": "veo_3_1_i2v_s_fast_ultra_relaxed",
173
+ "VIDEO_ASPECT_RATIO_PORTRAIT": "veo_3_1_i2v_s_fast_ultra_relaxed",
174
+ },
175
+ },
176
+ }
177
+
178
+ DEFAULT_VIDEO_QUALITY = "fast"
179
+
180
+
181
+ def resolve_video_model(
182
+ paygate_tier: str, aspect_ratio: str, quality: Optional[str] = None
183
+ ) -> Optional[str]:
184
+ """Resolve a Flow video model key from tier + aspect + quality.
185
+
186
+ Falls back through (quality → fast) → (tier → TIER_ONE) → None so
187
+ a stale frontend or unknown tier can't break dispatch silently.
188
+ """
189
+ q = (quality or DEFAULT_VIDEO_QUALITY).lower()
190
+ tier_map = (
191
+ VIDEO_MODEL_KEYS.get(paygate_tier)
192
+ or VIDEO_MODEL_KEYS.get("PAYGATE_TIER_ONE")
193
+ or {}
194
+ )
195
+ quality_map = tier_map.get(q) or tier_map.get(DEFAULT_VIDEO_QUALITY) or {}
196
+ return quality_map.get(aspect_ratio)
197
+
198
+ # project_id must match the shape Google Flow returns (UUID-ish). Validated at
199
+ # handler boundaries to prevent path traversal into arbitrary API URLs.
200
+ _PROJECT_ID_RE = re.compile(r"^[A-Za-z0-9_-]{1,128}$")
201
+
202
+ # Flow CDN URLs embed the UUID media_id in the path:
203
+ # https://flow-content.google/video/<UUID>?Expires=...&Signature=...
204
+ # When the polling response omits `metadata.video.mediaId` (it usually does for
205
+ # video — only `mediaGenerationId` which is a base64 protobuf, NOT a UUID),
206
+ # we recover the UUID from the URL exactly like flowkit does.
207
+ _UUID_IN_URL_RE = re.compile(
208
+ r"/([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})",
209
+ re.IGNORECASE,
210
+ )
211
+
212
+
213
+ def _media_id_from_url(url: Optional[str]) -> Optional[str]:
214
+ if not isinstance(url, str):
215
+ return None
216
+ m = _UUID_IN_URL_RE.search(url)
217
+ return m.group(1) if m else None
218
+
219
+
220
+ def _first_media_entry(resp: Any, mid: str) -> dict[str, Any]:
221
+ """Pull the media object for ``mid`` from a batchCheckAsync media response.
222
+
223
+ The body is either a bare ``[{...}]`` list or ``{media:[{...}]}`` — handle
224
+ both. Falls back to the first entry when the name doesn't match.
225
+ """
226
+ data = resp.get("data") if isinstance(resp, dict) else None
227
+ media = (
228
+ data if isinstance(data, list)
229
+ else (data.get("media") if isinstance(data, dict) else None)
230
+ )
231
+ if not isinstance(media, list):
232
+ return {}
233
+ for m in media:
234
+ if isinstance(m, dict) and m.get("name") == mid:
235
+ return m
236
+ return media[0] if media and isinstance(media[0], dict) else {}
237
+
238
+
239
+ def _media_url_from_entry(m: dict[str, Any]) -> Optional[str]:
240
+ """Best-effort playable URL from a finished media object (if Flow inlines
241
+ one in the status payload; otherwise the caller fetches the bytes)."""
242
+ video = m.get("video") if isinstance(m.get("video"), dict) else {}
243
+ gen = video.get("generatedVideo") if isinstance(video.get("generatedVideo"), dict) else {}
244
+ for cand in (
245
+ video.get("fifeUrl"), gen.get("fifeUrl"),
246
+ video.get("servingBaseUri"), gen.get("servingBaseUri"),
247
+ m.get("fifeUrl"),
248
+ ):
249
+ if isinstance(cand, str) and cand:
250
+ return cand
251
+ return None
252
+
253
+
254
+ def _extract_inner_api_error(resp: Any) -> Optional[str]:
255
+ """Surface a Flow API error if the response envelope indicates one.
256
+
257
+ `flow_client.api_request` returns ``{"id", "status", "data"}`` on a
258
+ completed round-trip even when the underlying call failed — the HTTP
259
+ status from Google lives on ``resp["status"]`` and the structured error
260
+ body on ``resp["data"]["error"]``. Top-level ``resp["error"]`` is only
261
+ set by the client itself for transport-level failures (which we already
262
+ handle). When ``status >= 400`` or ``data.error.status`` is set, the
263
+ request must be reported as failed — silently treating an empty
264
+ ``media_ids`` list as success masked Flow's content-filter rejections
265
+ (e.g. ``PUBLIC_ERROR_PROMINENT_PEOPLE_FILTER_FAILED``).
266
+ """
267
+ if not isinstance(resp, dict):
268
+ return None
269
+ status = resp.get("status")
270
+ data = resp.get("data") if isinstance(resp.get("data"), dict) else None
271
+ err = data.get("error") if isinstance(data, dict) else None
272
+ has_status_err = isinstance(status, int) and status >= 400
273
+ has_data_err = isinstance(err, dict)
274
+ if not (has_status_err or has_data_err):
275
+ return None
276
+ if has_data_err:
277
+ reasons: list[str] = []
278
+ for detail in err.get("details") or []:
279
+ if isinstance(detail, dict):
280
+ r = detail.get("reason")
281
+ if isinstance(r, str) and r:
282
+ reasons.append(r)
283
+ msg = err.get("message") or err.get("status") or "API error"
284
+ return f"{reasons[0]}: {msg}" if reasons else str(msg)
285
+ return f"API_{status}"
286
+
287
+
288
+ def is_valid_project_id(project_id: str) -> bool:
289
+ return bool(_PROJECT_ID_RE.fullmatch(project_id))
290
+
291
+ # Captcha action strings recognised by Google Flow.
292
+ CAPTCHA_IMAGE = "IMAGE_GENERATION"
293
+ CAPTCHA_VIDEO = "VIDEO_GENERATION"
294
+
295
+ # Default max operations to poll in parallel. Conservative; flowkit passes
296
+ # the full list at once.
297
+ _MAX_VIDEO_OPS = 4
298
+
299
+ # Image variants per dispatch are capped server-side as defence-in-depth — the
300
+ # UI clamps to 4 too. Any value above this is silently coerced down.
301
+ MAX_VARIANT_COUNT = 4
302
+
303
+ # Minimal static headers that have worked against labs.google in flowkit.
304
+ _TRPC_HEADERS = {
305
+ "content-type": "application/json",
306
+ "accept": "*/*",
307
+ }
308
+ _API_HEADERS = {
309
+ "content-type": "text/plain;charset=UTF-8",
310
+ "accept": "*/*",
311
+ "origin": "https://labs.google",
312
+ "referer": "https://labs.google/",
313
+ }
314
+
315
+
316
+ def _client_context(project_id: str, paygate_tier: str) -> dict:
317
+ """Skeleton clientContext — extension fills in recaptchaContext.token.
318
+
319
+ `paygate_tier` is REQUIRED (no default). Pre-v1.1.5 the default was
320
+ `"PAYGATE_TIER_ONE"` which silently downgraded Ultra users when any
321
+ upstream code path forgot to pass tier. Now we raise loudly on
322
+ invalid / unknown values so a code regression can't quietly serve
323
+ Pro to an Ultra account.
324
+ """
325
+ if paygate_tier not in _VALID_TIERS:
326
+ raise ValueError(
327
+ f"invalid paygate_tier {paygate_tier!r} — must be one of {sorted(_VALID_TIERS)}"
328
+ )
329
+ return {
330
+ "projectId": str(project_id),
331
+ "recaptchaContext": {
332
+ "applicationType": "RECAPTCHA_APPLICATION_TYPE_WEB",
333
+ "token": "",
334
+ },
335
+ "sessionId": f";{int(time.time() * 1000)}",
336
+ "tool": "PINHOLE",
337
+ "userPaygateTier": paygate_tier,
338
+ }
339
+
340
+
341
+ def _generate_images_url(project_id: str) -> str:
342
+ return f"{FLOW_API_BASE}/v1/projects/{project_id}/flowMedia:batchGenerateImages"
343
+
344
+
345
+ class FlowSDK:
346
+ """High-level helpers on top of a FlowClient. Stateless; one per request."""
347
+
348
+ def __init__(self, client: Optional[FlowClient] = None) -> None:
349
+ if client is None:
350
+ raise ValueError(
351
+ "FlowSDK requires an explicit FlowClient — the module-level "
352
+ "singleton has been removed; obtain a client via pool.get_by_api_key()."
353
+ )
354
+ self._client = client
355
+
356
+ # ── project listing (TRPC search) ──────────────────────────────────────
357
+ async def search_user_projects(
358
+ self,
359
+ cursor: Optional[str] = None,
360
+ page_size: int = 20,
361
+ tool: str = "PINHOLE",
362
+ ) -> dict[str, Any]:
363
+ """Fetch one page of the user's Flow project list.
364
+
365
+ Returns ``{raw, projects, next_page_token}`` on success or
366
+ ``{raw, error}`` on failure. Each project is a dict with
367
+ ``project_id`` and ``project_title`` plus optional
368
+ ``thumbnail_media_key`` and ``creation_time``.
369
+ """
370
+ import json as _json
371
+ from urllib.parse import quote
372
+
373
+ input_json: dict[str, Any] = {
374
+ "json": {
375
+ "pageSize": page_size,
376
+ "toolName": tool,
377
+ "cursor": cursor,
378
+ },
379
+ }
380
+ # TRPC encodes the literal `undefined` cursor via meta on the
381
+ # first call. Subsequent calls pass a string cursor directly.
382
+ if cursor is None:
383
+ input_json["meta"] = {"values": {"cursor": ["undefined"]}}
384
+
385
+ url = (
386
+ f"{TRPC_SEARCH_PROJECTS}?input="
387
+ + quote(_json.dumps(input_json, separators=(",", ":")))
388
+ )
389
+ resp = await self._client.trpc_request(
390
+ url=url,
391
+ method="GET",
392
+ headers=_TRPC_HEADERS,
393
+ body=None,
394
+ )
395
+ if isinstance(resp, dict) and resp.get("error"):
396
+ return {"raw": resp, "error": resp["error"]}
397
+
398
+ # Response shape (per the canonical TRPC envelope):
399
+ # data.result.data.json.result.{projects, nextPageToken}
400
+ try:
401
+ inner = resp["data"]["result"]["data"]["json"]["result"]
402
+ except (KeyError, TypeError):
403
+ return {"raw": resp, "error": "unexpected_search_response_shape"}
404
+
405
+ raw_projects = inner.get("projects") or []
406
+ projects: list[dict[str, Any]] = []
407
+ for p in raw_projects:
408
+ if not isinstance(p, dict):
409
+ continue
410
+ pid = p.get("projectId")
411
+ if not isinstance(pid, str) or not pid:
412
+ continue
413
+ info = p.get("projectInfo") if isinstance(p.get("projectInfo"), dict) else {}
414
+ projects.append({
415
+ "project_id": pid,
416
+ "project_title": info.get("projectTitle") or "Untitled",
417
+ "thumbnail_media_key": info.get("thumbnailMediaKey"),
418
+ "creation_time": p.get("creationTime"),
419
+ })
420
+ return {
421
+ "raw": resp,
422
+ "projects": projects,
423
+ "next_page_token": inner.get("nextPageToken"),
424
+ }
425
+
426
+ async def list_user_projects_all(
427
+ self, tool: str = "PINHOLE", max_pages: int = 10
428
+ ) -> dict[str, Any]:
429
+ """Paginate `search_user_projects` until exhausted (or max_pages).
430
+
431
+ Returns ``{projects, truncated}`` — truncated is True when we hit
432
+ the page cap with a non-null next_page_token still present.
433
+ """
434
+ all_projects: list[dict[str, Any]] = []
435
+ cursor: Optional[str] = None
436
+ for _ in range(max_pages):
437
+ page = await self.search_user_projects(cursor=cursor, tool=tool)
438
+ if page.get("error"):
439
+ return {
440
+ "projects": all_projects,
441
+ "truncated": True,
442
+ "error": page["error"],
443
+ }
444
+ all_projects.extend(page.get("projects") or [])
445
+ cursor = page.get("next_page_token")
446
+ if not cursor:
447
+ return {"projects": all_projects, "truncated": False}
448
+ return {"projects": all_projects, "truncated": True}
449
+
450
+ # ── project creation (TRPC) ────────────────────────────────────────────
451
+ async def create_project(
452
+ self, title: str, tool: str = "PINHOLE"
453
+ ) -> dict[str, Any]:
454
+ body = {"json": {"projectTitle": title, "toolName": tool}}
455
+ resp = await self._client.trpc_request(
456
+ url=TRPC_CREATE_PROJECT,
457
+ method="POST",
458
+ headers=_TRPC_HEADERS,
459
+ body=body,
460
+ )
461
+ if isinstance(resp, dict) and resp.get("error"):
462
+ return {"raw": resp, "error": resp["error"]}
463
+
464
+ project_id = _extract_project_id(resp)
465
+ out: dict[str, Any] = {"raw": resp}
466
+ if project_id is None:
467
+ out["error"] = "no_project_id_in_response"
468
+ else:
469
+ out["project_id"] = project_id
470
+ return out
471
+
472
+ # ── model catalog (TRPC) ───────────────────────────────────────────────
473
+ async def get_project_initial_data(self, project_id: str) -> dict[str, Any]:
474
+ """Fetch the project's initial data, which carries the full model
475
+ catalog. Returns ``{raw, data}`` on success (``data`` is the inner
476
+ ``json`` block with ``modelConfig`` / ``userData`` /
477
+ ``projectContents.externalReferenceMedia``) or ``{raw, error}``.
478
+ """
479
+ import json as _json
480
+ from urllib.parse import quote
481
+
482
+ input_json = {"json": {"projectId": str(project_id)}}
483
+ url = (
484
+ f"{TRPC_PROJECT_INITIAL_DATA}?input="
485
+ + quote(_json.dumps(input_json, separators=(",", ":")))
486
+ )
487
+ resp = await self._client.trpc_request(
488
+ url=url, method="GET", headers=_TRPC_HEADERS, body=None
489
+ )
490
+ if isinstance(resp, dict) and resp.get("error"):
491
+ return {"raw": resp, "error": resp["error"]}
492
+ try:
493
+ data = resp["data"]["result"]["data"]["json"]
494
+ except (KeyError, TypeError):
495
+ return {"raw": resp, "error": "unexpected_initial_data_shape"}
496
+ return {"raw": resp, "data": data}
497
+
498
+ # ── video generation (async via operations) ────────────────────────────
499
+ async def gen_video(
500
+ self,
501
+ prompt: str,
502
+ project_id: str,
503
+ start_media_id: Optional[str] = None,
504
+ aspect_ratio: str = "VIDEO_ASPECT_RATIO_LANDSCAPE",
505
+ paygate_tier: Optional[str] = None,
506
+ scene_id: Optional[str] = None,
507
+ start_media_ids: Optional[list[str]] = None,
508
+ video_quality: Optional[str] = None,
509
+ model_key: Optional[str] = None,
510
+ ) -> dict[str, Any]:
511
+ """Kick off i2v operation(s). Returns ``{raw, operation_names}`` on
512
+ success or ``{raw, error}`` on failure. Operations are async — the
513
+ caller polls ``check_async`` until they complete.
514
+
515
+ ``start_media_ids`` (optional list) — when provided, dispatch ONE
516
+ item per source image so a 4-variant upstream image produces 4
517
+ videos in a single batch (one operation per source). Falls back to
518
+ ``start_media_id`` (single) if the list is missing/empty.
519
+
520
+ ``video_quality`` ("fast" / "lite" / "quality" / "lite_relaxed"
521
+ / "fast_relaxed") routes to a different Veo checkpoint. Defaults
522
+ to "fast" — the `_s_fast` family. The first three are available
523
+ on both Tier 1 (Pro) and Tier 2 (Ultra); the `_relaxed` variants
524
+ are 0-credit low-priority queues and are Ultra-only. See
525
+ ``VIDEO_MODEL_KEYS`` for the per-tier mapping.
526
+
527
+ ``paygate_tier`` is required. Pre-v1.1.5 it defaulted to
528
+ ``"PAYGATE_TIER_ONE"`` which silently downgraded Ultra users.
529
+ Raise loudly instead — the worker should always have a tier
530
+ from the live extension signal before reaching here.
531
+ """
532
+ if paygate_tier is None:
533
+ raise ValueError("paygate_tier is required — see docs/migrations/clear-polluted-paygate-tier.sql")
534
+ # Prefer a catalog-resolved key (dynamic, tier-correct); fall back to
535
+ # the static tier/quality/aspect map only when none was supplied.
536
+ if not model_key:
537
+ model_key = resolve_video_model(paygate_tier, aspect_ratio, video_quality)
538
+ if not model_key:
539
+ return {
540
+ "raw": None,
541
+ "error": (
542
+ f"no_video_model_for_tier_{paygate_tier}"
543
+ f"_quality_{video_quality or DEFAULT_VIDEO_QUALITY}"
544
+ f"_aspect_{aspect_ratio}"
545
+ ),
546
+ }
547
+
548
+ # Normalise into a non-empty list of source media ids. Single
549
+ # `start_media_id` is the common case; `start_media_ids` is for
550
+ # batch-i2v from a multi-variant upstream image.
551
+ sources: list[str] = []
552
+ if start_media_ids:
553
+ sources = [m for m in start_media_ids if isinstance(m, str) and m]
554
+ if not sources and isinstance(start_media_id, str) and start_media_id:
555
+ sources = [start_media_id]
556
+ if not sources:
557
+ return {"raw": None, "error": "missing_start_media_id"}
558
+
559
+ ts = int(time.time() * 1000)
560
+ ctx = _client_context(project_id, paygate_tier)
561
+ items: list[dict[str, Any]] = []
562
+ for i, mid in enumerate(sources):
563
+ items.append({
564
+ "aspectRatio": aspect_ratio,
565
+ # Distinct seed per item so Flow doesn't dedupe.
566
+ "seed": (ts + i * 9973) % 1_000_000,
567
+ "textInput": {"structuredPrompt": {"parts": [{"text": prompt}]}},
568
+ "videoModelKey": model_key,
569
+ "startImage": {"mediaId": mid},
570
+ "metadata": {"sceneId": scene_id or str(uuid.uuid4())},
571
+ })
572
+ body = {
573
+ "clientContext": ctx,
574
+ "mediaGenerationContext": {"batchId": str(uuid.uuid4())},
575
+ "requests": items,
576
+ "useV2ModelConfig": True,
577
+ }
578
+
579
+ resp = await self._client.api_request(
580
+ url=VIDEO_I2V_URL,
581
+ method="POST",
582
+ headers=dict(_API_HEADERS),
583
+ body=body,
584
+ captcha_action=CAPTCHA_VIDEO,
585
+ )
586
+ if isinstance(resp, dict) and resp.get("error"):
587
+ return {"raw": resp, "error": resp["error"]}
588
+ inner_err = _extract_inner_api_error(resp)
589
+ if inner_err:
590
+ return {"raw": resp, "error": inner_err}
591
+
592
+ op_names = extract_operation_names(resp)
593
+ if not op_names:
594
+ return {"raw": resp, "error": "no_operations_in_response"}
595
+ out: dict[str, Any] = {"raw": resp, "operation_names": op_names}
596
+ # NEW low-priority workflow models return `data.workflows[]` with a
597
+ # `primaryMediaId` per workflow instead of operations. Surface the
598
+ # pairing so the poller can hit `/v1/media/<id>` directly.
599
+ workflows = extract_video_workflows(resp)
600
+ if workflows:
601
+ out["workflows"] = workflows
602
+ return out
603
+
604
+ # ── Veo 3.1 reference-to-video (r2v) ───────────────────────────────────
605
+ async def gen_video_r2v(
606
+ self,
607
+ prompt: str,
608
+ project_id: str,
609
+ ref_media_ids: list[str],
610
+ aspect_ratio: str = "VIDEO_ASPECT_RATIO_PORTRAIT",
611
+ paygate_tier: Optional[str] = None,
612
+ seed: Optional[int] = None,
613
+ reference_audio_ids: Optional[list[str]] = None,
614
+ model_key: Optional[str] = None,
615
+ scene_id: Optional[str] = None,
616
+ ) -> dict[str, Any]:
617
+ """Kick off Veo 3.1 reference-to-video generation. Uses
618
+ /video:batchAsyncGenerateVideoReferenceImages with a referenceImages[]
619
+ payload — distinct from the (now-stale) i2v startImage path.
620
+
621
+ Returns ``{raw, operation_names}`` (+ ``workflows`` for the
622
+ low-priority workflow schema) on success or ``{raw, error}`` on
623
+ failure. Polls via the shared ``check_async`` path.
624
+
625
+ ``ref_media_ids`` MUST be non-empty (the model is
626
+ reference-conditioned, not text-only).
627
+ ``aspect_ratio`` ∈ {PORTRAIT, LANDSCAPE}.
628
+ ``reference_audio_ids`` — optional audio reference media ids (e.g.
629
+ ``"algenib"``). When provided, attached as ``referenceAudio``; omitted
630
+ entirely otherwise.
631
+ ``paygate_tier`` required — same as gen_video.
632
+ """
633
+ if paygate_tier is None:
634
+ raise ValueError("paygate_tier is required")
635
+ if aspect_ratio not in R2V_VALID_ASPECTS:
636
+ return {
637
+ "raw": None,
638
+ "error": f"r2v_aspect_unsupported_{aspect_ratio}",
639
+ }
640
+ cleaned_refs = [m for m in (ref_media_ids or []) if isinstance(m, str) and m]
641
+ if not cleaned_refs:
642
+ return {"raw": None, "error": "missing_ref_media_ids"}
643
+
644
+ ts = int(time.time() * 1000)
645
+ used_seed = seed if seed is not None else ts % 1_000_000
646
+ ctx = _client_context(project_id, paygate_tier)
647
+ request_item: dict[str, Any] = {
648
+ "aspectRatio": aspect_ratio,
649
+ "textInput": {"structuredPrompt": {"parts": [{"text": prompt}]}},
650
+ "videoModelKey": model_key or VEO_R2V_MODEL_KEY,
651
+ "seed": used_seed,
652
+ "metadata": {"sceneId": scene_id} if scene_id else {},
653
+ "referenceImages": [
654
+ {"mediaId": mid, "imageUsageType": "IMAGE_USAGE_TYPE_ASSET"}
655
+ for mid in cleaned_refs
656
+ ],
657
+ }
658
+ cleaned_audio = [a for a in (reference_audio_ids or []) if isinstance(a, str) and a]
659
+ if cleaned_audio:
660
+ request_item["referenceAudio"] = [{"mediaId": a} for a in cleaned_audio]
661
+ body = {
662
+ "mediaGenerationContext": {
663
+ "batchId": str(uuid.uuid4()),
664
+ # V2 config flags silent-audio outputs as failures so the
665
+ # caller can retry instead of getting a degraded video.
666
+ "audioFailurePreference": "BLOCK_SILENCED_VIDEOS",
667
+ },
668
+ "clientContext": {**ctx, "sessionId": f";{ts}"},
669
+ "requests": [request_item],
670
+ "useV2ModelConfig": True,
671
+ }
672
+
673
+ resp = await self._client.api_request(
674
+ url=VIDEO_R2V_URL,
675
+ method="POST",
676
+ headers=dict(_API_HEADERS),
677
+ body=body,
678
+ captcha_action=CAPTCHA_VIDEO,
679
+ )
680
+ if isinstance(resp, dict) and resp.get("error"):
681
+ return {"raw": resp, "error": resp["error"]}
682
+ inner_err = _extract_inner_api_error(resp)
683
+ if inner_err:
684
+ return {"raw": resp, "error": inner_err}
685
+
686
+ op_names = extract_operation_names(resp)
687
+ if not op_names:
688
+ return {"raw": resp, "error": "no_operations_in_response"}
689
+ out: dict[str, Any] = {"raw": resp, "operation_names": op_names}
690
+ workflows = extract_video_workflows(resp)
691
+ if workflows:
692
+ out["workflows"] = workflows
693
+ if scene_id:
694
+ out["scene_id"] = scene_id
695
+ return out
696
+
697
+ # ── Veo 3.1 text-to-video (t2v) ────────────────────────────────────────
698
+ async def gen_video_t2v(
699
+ self,
700
+ prompt: str,
701
+ project_id: str,
702
+ aspect_ratio: str = "VIDEO_ASPECT_RATIO_LANDSCAPE",
703
+ paygate_tier: Optional[str] = None,
704
+ seed: Optional[int] = None,
705
+ model_key: Optional[str] = None,
706
+ scene_id: Optional[str] = None,
707
+ ) -> dict[str, Any]:
708
+ """Kick off text-to-video. Same workflow path as r2v but with no image
709
+ or audio inputs — hits /video:batchAsyncGenerateVideoText.
710
+
711
+ ``model_key`` is required (resolved from the catalog) — there is no
712
+ static fallback for t2v.
713
+
714
+ ``scene_id`` (optional) pins the output into a client-chosen scene via
715
+ ``metadata.sceneId`` — Flow otherwise auto-creates a scene and never
716
+ tells us its id, which would leave the clip un-extendable. When set, the
717
+ scene id is echoed back in the result so the caller can remember it.
718
+ """
719
+ if paygate_tier is None:
720
+ raise ValueError("paygate_tier is required")
721
+ if aspect_ratio not in R2V_VALID_ASPECTS:
722
+ return {"raw": None, "error": f"t2v_aspect_unsupported_{aspect_ratio}"}
723
+ if not model_key:
724
+ return {"raw": None, "error": "t2v_model_key_required"}
725
+
726
+ ts = int(time.time() * 1000)
727
+ used_seed = seed if seed is not None else ts % 1_000_000
728
+ ctx = _client_context(project_id, paygate_tier)
729
+ request_item = {
730
+ "aspectRatio": aspect_ratio,
731
+ "textInput": {"structuredPrompt": {"parts": [{"text": prompt}]}},
732
+ "videoModelKey": model_key,
733
+ "seed": used_seed,
734
+ "metadata": {"sceneId": scene_id} if scene_id else {},
735
+ }
736
+ body = {
737
+ "mediaGenerationContext": {
738
+ "batchId": str(uuid.uuid4()),
739
+ "audioFailurePreference": "BLOCK_SILENCED_VIDEOS",
740
+ },
741
+ "clientContext": {**ctx, "sessionId": f";{ts}"},
742
+ "requests": [request_item],
743
+ "useV2ModelConfig": True,
744
+ }
745
+
746
+ resp = await self._client.api_request(
747
+ url=VIDEO_T2V_URL,
748
+ method="POST",
749
+ headers=dict(_API_HEADERS),
750
+ body=body,
751
+ captcha_action=CAPTCHA_VIDEO,
752
+ )
753
+ if isinstance(resp, dict) and resp.get("error"):
754
+ return {"raw": resp, "error": resp["error"]}
755
+ inner_err = _extract_inner_api_error(resp)
756
+ if inner_err:
757
+ return {"raw": resp, "error": inner_err}
758
+
759
+ op_names = extract_operation_names(resp)
760
+ if not op_names:
761
+ return {"raw": resp, "error": "no_operations_in_response"}
762
+ out: dict[str, Any] = {"raw": resp, "operation_names": op_names}
763
+ workflows = extract_video_workflows(resp)
764
+ if workflows:
765
+ out["workflows"] = workflows
766
+ if scene_id:
767
+ out["scene_id"] = scene_id
768
+ return out
769
+
770
+ # ── Veo 3.1 extend (continue a clip) ───────────────────────────────────
771
+ async def gen_video_extend(
772
+ self,
773
+ prompt: str,
774
+ project_id: str,
775
+ source_media_id: str,
776
+ scene_id: str,
777
+ position: int = 1,
778
+ aspect_ratio: str = "VIDEO_ASPECT_RATIO_LANDSCAPE",
779
+ paygate_tier: Optional[str] = None,
780
+ seed: Optional[int] = None,
781
+ model_key: Optional[str] = None,
782
+ ) -> dict[str, Any]:
783
+ """Extend an existing Veo clip (``batchAsyncGenerateVideoExtendVideo``).
784
+
785
+ ``source_media_id`` is the clip to continue — for extend this is the
786
+ source video's *operation/generation id* (``video.operation.name``),
787
+ NOT its workflow primaryMediaId. ``scene_id`` is the scene the source
788
+ belongs to and ``position`` is the slot the extension occupies in that
789
+ scene's timeline (1 for the first extension). Only Veo-generated videos
790
+ can be extended.
791
+
792
+ Returns the workflow-schema shape (``{raw, operation_names, workflows}``)
793
+ like the other video dispatchers; poll via the shared ``check_async``.
794
+ """
795
+ if paygate_tier is None:
796
+ raise ValueError("paygate_tier is required")
797
+ if not model_key:
798
+ return {"raw": None, "error": "extend_model_key_required"}
799
+ if not isinstance(source_media_id, str) or not source_media_id:
800
+ return {"raw": None, "error": "missing_source_media_id"}
801
+ if not isinstance(scene_id, str) or not scene_id:
802
+ return {"raw": None, "error": "missing_scene_id"}
803
+
804
+ ts = int(time.time() * 1000)
805
+ used_seed = seed if seed is not None else ts % 1_000_000
806
+ ctx = _client_context(project_id, paygate_tier)
807
+ body = {
808
+ "mediaGenerationContext": {
809
+ "batchId": str(uuid.uuid4()),
810
+ "audioFailurePreference": "BLOCK_SILENCED_VIDEOS",
811
+ "sceneContext": {"sceneId": scene_id, "position": position},
812
+ },
813
+ "clientContext": {**ctx, "sessionId": f";{ts}"},
814
+ "requests": [
815
+ {
816
+ "aspectRatio": aspect_ratio,
817
+ "textInput": {"structuredPrompt": {"parts": [{"text": prompt}]}},
818
+ "videoModelKey": model_key,
819
+ "seed": used_seed,
820
+ "metadata": {"sceneId": scene_id},
821
+ "videoInput": {"mediaId": source_media_id},
822
+ }
823
+ ],
824
+ "useV2ModelConfig": True,
825
+ }
826
+ return await self._dispatch_video(VIDEO_EXTEND_URL, body)
827
+
828
+ # ── Veo 3.1 edit (video-to-video) ──────────────────────────────────────
829
+ async def gen_video_edit(
830
+ self,
831
+ prompt: str,
832
+ project_id: str,
833
+ source_media_id: str,
834
+ workflow_id: str,
835
+ end_frame_index: int,
836
+ start_frame_index: int = 0,
837
+ aspect_ratio: str = "VIDEO_ASPECT_RATIO_LANDSCAPE",
838
+ paygate_tier: Optional[str] = None,
839
+ seed: Optional[int] = None,
840
+ model_key: Optional[str] = None,
841
+ ) -> dict[str, Any]:
842
+ """Edit an existing video (``batchAsyncGenerateVideoEditVideo``, v2v).
843
+
844
+ ``source_media_id`` is the source video's workflow primaryMediaId (its
845
+ media ``name``). ``workflow_id`` is the source's workflow; the edit
846
+ produces a new media version under it. ``start_frame_index`` /
847
+ ``end_frame_index`` select the slice to edit — frames are 30fps, so a
848
+ full 8s clip is 0..240. Verified key: ``abra_edit``.
849
+
850
+ Returns the workflow-schema shape; poll via the shared ``check_async``.
851
+ """
852
+ if paygate_tier is None:
853
+ raise ValueError("paygate_tier is required")
854
+ if not model_key:
855
+ return {"raw": None, "error": "edit_model_key_required"}
856
+ if not isinstance(source_media_id, str) or not source_media_id:
857
+ return {"raw": None, "error": "missing_source_media_id"}
858
+ if not isinstance(workflow_id, str) or not workflow_id:
859
+ return {"raw": None, "error": "missing_workflow_id"}
860
+
861
+ ts = int(time.time() * 1000)
862
+ used_seed = seed if seed is not None else ts % 1_000_000
863
+ ctx = _client_context(project_id, paygate_tier)
864
+ body = {
865
+ "mediaGenerationContext": {
866
+ "batchId": str(uuid.uuid4()),
867
+ "audioFailurePreference": "BLOCK_SILENCED_VIDEOS",
868
+ },
869
+ "clientContext": {**ctx, "sessionId": f";{ts}"},
870
+ "requests": [
871
+ {
872
+ "aspectRatio": aspect_ratio,
873
+ "textInput": {"structuredPrompt": {"parts": [{"text": prompt}]}},
874
+ "videoModelKey": model_key,
875
+ "seed": used_seed,
876
+ "metadata": {"workflowId": workflow_id},
877
+ "videoInput": {
878
+ "mediaId": source_media_id,
879
+ "startFrameIndex": start_frame_index,
880
+ "endFrameIndex": end_frame_index,
881
+ },
882
+ }
883
+ ],
884
+ }
885
+ return await self._dispatch_video(VIDEO_EDIT_URL, body)
886
+
887
+ async def resolve_scene(
888
+ self, project_id: str, workflow_ids: list[str]
889
+ ) -> dict[str, Any]:
890
+ """Resolve the scene that holds a set of workflows, via
891
+ ``POST /v1/flow/projects/{id}/scenes`` with ``{workflowIds:[...]}``.
892
+
893
+ This is how the web client learns a clip's scene id before extending it
894
+ — Flow places every generated video in an auto-created scene but never
895
+ echoes the id anywhere else. Returns ``{raw, scene_id, position}`` where
896
+ ``position`` is the next free slot in the scene's timeline (the count of
897
+ workflows already in it), or ``{raw, error}``.
898
+ """
899
+ cleaned = [w for w in workflow_ids if isinstance(w, str) and w]
900
+ if not cleaned:
901
+ return {"raw": None, "error": "missing_workflow_ids"}
902
+ url = f"{FLOW_API_BASE}/v1/flow/projects/{project_id}/scenes"
903
+ resp = await self._client.api_request(
904
+ url=url,
905
+ method="POST",
906
+ headers=dict(_API_HEADERS),
907
+ body={"workflowIds": cleaned},
908
+ )
909
+ if isinstance(resp, dict) and resp.get("error"):
910
+ return {"raw": resp, "error": resp["error"]}
911
+ inner_err = _extract_inner_api_error(resp)
912
+ if inner_err:
913
+ return {"raw": resp, "error": inner_err}
914
+ data = resp.get("data") if isinstance(resp, dict) else None
915
+ scene = data.get("scene") if isinstance(data, dict) else None
916
+ scene_id = scene.get("sceneId") if isinstance(scene, dict) else None
917
+ if not isinstance(scene_id, str) or not scene_id:
918
+ return {"raw": resp, "error": "no_scene_id_in_response"}
919
+ sws = data.get("sceneWorkflows") if isinstance(data, dict) else None
920
+ position = len(sws) if isinstance(sws, list) else 1
921
+ return {"raw": resp, "scene_id": scene_id, "position": position}
922
+
923
+ async def fetch_video_meta(
924
+ self, media_ids: list[str], project_id: str
925
+ ) -> dict[str, Any]:
926
+ """Fetch rich metadata for finished video media via the media-based
927
+ ``batchCheckAsync`` (``{media:[{name,projectId}]}``) — distinct from the
928
+ operation-based poll. This is the only call that returns the workflow id,
929
+ operation/generation id, scene id, and dimensions needed to drive
930
+ extend/edit. Returns ``{raw, meta: {media_id: {...}}}``.
931
+ """
932
+ cleaned = [m for m in media_ids if isinstance(m, str) and m]
933
+ if not cleaned:
934
+ return {"raw": None, "meta": {}}
935
+ body = {"media": [{"name": m, "projectId": project_id} for m in cleaned]}
936
+ resp = await self._client.api_request(
937
+ url=VIDEO_POLL_URL,
938
+ method="POST",
939
+ headers=dict(_API_HEADERS),
940
+ body=body,
941
+ )
942
+ meta: dict[str, dict[str, Any]] = {}
943
+ data = resp.get("data") if isinstance(resp, dict) else None
944
+ media = data.get("media") if isinstance(data, dict) else None
945
+ for m in media or []:
946
+ if not isinstance(m, dict):
947
+ continue
948
+ name = m.get("name")
949
+ if not isinstance(name, str):
950
+ continue
951
+ video = m.get("video") if isinstance(m.get("video"), dict) else {}
952
+ op = video.get("operation") if isinstance(video.get("operation"), dict) else {}
953
+ dims = video.get("dimensions") if isinstance(video.get("dimensions"), dict) else {}
954
+ meta[name] = {
955
+ "workflow_id": m.get("workflowId"),
956
+ "operation_id": op.get("name"),
957
+ "scene_id": m.get("sceneId") or m.get("sceneIdContext"),
958
+ "length": dims.get("length"),
959
+ }
960
+ return {"raw": resp, "meta": meta}
961
+
962
+ async def _dispatch_video(self, url: str, body: dict[str, Any]) -> dict[str, Any]:
963
+ """Shared submit + response-shaping for the workflow-schema video
964
+ dispatchers (extend / edit). Sends the captcha-gated request and
965
+ returns ``{raw, operation_names, workflows}`` or ``{raw, error}``."""
966
+ resp = await self._client.api_request(
967
+ url=url,
968
+ method="POST",
969
+ headers=dict(_API_HEADERS),
970
+ body=body,
971
+ captcha_action=CAPTCHA_VIDEO,
972
+ )
973
+ if isinstance(resp, dict) and resp.get("error"):
974
+ return {"raw": resp, "error": resp["error"]}
975
+ inner_err = _extract_inner_api_error(resp)
976
+ if inner_err:
977
+ return {"raw": resp, "error": inner_err}
978
+ op_names = extract_operation_names(resp)
979
+ if not op_names:
980
+ return {"raw": resp, "error": "no_operations_in_response"}
981
+ out: dict[str, Any] = {"raw": resp, "operation_names": op_names}
982
+ workflows = extract_video_workflows(resp)
983
+ if workflows:
984
+ out["workflows"] = workflows
985
+ return out
986
+
987
+ async def check_async(
988
+ self,
989
+ operation_names: list[str],
990
+ workflows: Optional[list[dict[str, Any]]] = None,
991
+ project_id: Optional[str] = None,
992
+ ) -> dict[str, Any]:
993
+ """Poll one or more video operations. No captcha.
994
+
995
+ Returns ``{raw, operations: [{name, done, media_entries}]}`` — one
996
+ entry per input operation. ``media_entries`` is a list of
997
+ ``{media_id, url, mediaType}`` ready for ``media.ingest_urls``.
998
+
999
+ ``workflows`` (optional) carries ``{name, primary_media_id}`` pairs
1000
+ from the NEW low-priority response. When provided, every workflow
1001
+ entry is polled against ``/v1/media/<primary_media_id>`` and the
1002
+ result is merged into the same ``operations`` shape so the caller
1003
+ is schema-agnostic.
1004
+ """
1005
+ ops_summary: list[dict[str, Any]] = []
1006
+ raw_old: Any = None
1007
+ # Names that came from workflows are NOT valid operation handles —
1008
+ # don't dispatch them to batchCheckAsync (Flow would 400).
1009
+ workflow_names = {w["name"] for w in (workflows or []) if isinstance(w, dict) and w.get("name")}
1010
+ old_names = [n for n in operation_names if n not in workflow_names]
1011
+ if old_names:
1012
+ body = {
1013
+ "operations": [
1014
+ {"operation": {"name": name}} for name in old_names
1015
+ ]
1016
+ }
1017
+ raw_old = await self._client.api_request(
1018
+ url=VIDEO_POLL_URL,
1019
+ method="POST",
1020
+ headers=dict(_API_HEADERS),
1021
+ body=body,
1022
+ )
1023
+ if isinstance(raw_old, dict) and raw_old.get("error"):
1024
+ return {"raw": raw_old, "error": raw_old["error"]}
1025
+ ops_summary.extend(
1026
+ extract_video_operations(raw_old, requested=old_names)
1027
+ )
1028
+
1029
+ raw_workflows: list[dict[str, Any]] = []
1030
+ if workflows:
1031
+ wf_summary, raw_workflows = await self._poll_workflows(
1032
+ workflows, project_id
1033
+ )
1034
+ ops_summary.extend(wf_summary)
1035
+
1036
+ # Preserve the original input order so callers (worker) can keep
1037
+ # positional alignment with their per-op state.
1038
+ order = {name: i for i, name in enumerate(operation_names)}
1039
+ ops_summary.sort(key=lambda op: order.get(op.get("name"), 1 << 30))
1040
+
1041
+ raw_out: dict[str, Any] = {}
1042
+ if raw_old is not None:
1043
+ raw_out["operations_poll"] = raw_old
1044
+ if raw_workflows:
1045
+ raw_out["workflow_polls"] = raw_workflows
1046
+ return {"raw": raw_out or raw_old, "operations": ops_summary}
1047
+
1048
+ async def _poll_workflows(
1049
+ self,
1050
+ workflows: list[dict[str, Any]],
1051
+ project_id: Optional[str] = None,
1052
+ ) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
1053
+ """Single poll pass for workflow-mode (Low Priority) submissions.
1054
+
1055
+ Flow's web client checks these via
1056
+ ``POST /v1/video:batchCheckAsyncVideoGenerationStatus`` with a
1057
+ ``{media:[{name, projectId}]}`` body and reads
1058
+ ``mediaMetadata.mediaStatus.mediaGenerationStatus``
1059
+ (SCHEDULED → ACTIVE → SUCCESSFUL). The older ``GET /v1/media/<id>``
1060
+ status probe now returns ``400 INVALID_ARGUMENT`` while the media is
1061
+ still scheduled, which aborted otherwise-fine jobs — so status MUST
1062
+ come from the batch-check call. Only once the media is SUCCESSFUL do
1063
+ we fetch the encoded MP4 bytes (``GET /v1/media/<id>`` → base64).
1064
+
1065
+ Returns ``(ops_summary, raw_polls)`` mirroring the OLD-schema
1066
+ ``check_async`` contract: one entry per workflow with
1067
+ ``{name, done, media_entries, status, error}``. The poll loop in
1068
+ the worker calls this repeatedly via ``check_async`` until ``done``.
1069
+ """
1070
+ ops_summary: list[dict[str, Any]] = []
1071
+ raw_polls: list[dict[str, Any]] = []
1072
+ for wf in workflows:
1073
+ if not isinstance(wf, dict):
1074
+ continue
1075
+ name = wf.get("name")
1076
+ mid = wf.get("primary_media_id")
1077
+ if not isinstance(name, str) or not isinstance(mid, str) or not mid:
1078
+ continue
1079
+ pending = {
1080
+ "name": name, "done": False, "media_entries": [],
1081
+ "status": None, "error": None,
1082
+ }
1083
+ # 1. Status via the media-based batchCheckAsync (same call the web
1084
+ # client uses). projectId is required by Flow.
1085
+ try:
1086
+ status_resp = await self._client.api_request(
1087
+ url=VIDEO_POLL_URL,
1088
+ method="POST",
1089
+ headers=dict(_API_HEADERS),
1090
+ body={"media": [{"name": mid, "projectId": project_id}]},
1091
+ )
1092
+ except Exception as exc: # noqa: BLE001
1093
+ logger.warning("workflow status poll error for %s: %s", mid[:8], exc)
1094
+ ops_summary.append(pending)
1095
+ continue
1096
+ raw_polls.append({"name": name, "media_id": mid, "resp": status_resp})
1097
+
1098
+ m = _first_media_entry(status_resp, mid)
1099
+ m_status = (
1100
+ ((m.get("mediaMetadata") or {}).get("mediaStatus") or {})
1101
+ .get("mediaGenerationStatus")
1102
+ )
1103
+ if m_status == "MEDIA_GENERATION_STATUS_FAILED":
1104
+ ops_summary.append({
1105
+ **pending, "done": True, "status": m_status,
1106
+ "error": "MEDIA_GENERATION_STATUS_FAILED",
1107
+ })
1108
+ continue
1109
+ if m_status != "MEDIA_GENERATION_STATUS_SUCCESSFUL":
1110
+ # SCHEDULED / ACTIVE / PENDING / unknown — keep polling.
1111
+ ops_summary.append(pending)
1112
+ continue
1113
+
1114
+ # 2. SUCCESSFUL. Prefer a direct URL if the status payload carries
1115
+ # one; otherwise fetch the encoded MP4 bytes.
1116
+ url = _media_url_from_entry(m)
1117
+ if url:
1118
+ ops_summary.append({
1119
+ "name": name, "done": True,
1120
+ "media_entries": [
1121
+ {"media_id": mid, "url": url, "mediaType": "video"}
1122
+ ],
1123
+ "status": m_status, "error": None,
1124
+ })
1125
+ continue
1126
+ entry = await self._fetch_workflow_bytes(mid)
1127
+ if entry is None:
1128
+ # Status is SUCCESSFUL but bytes not materialised yet — keep
1129
+ # polling rather than failing.
1130
+ ops_summary.append(pending)
1131
+ continue
1132
+ ops_summary.append({
1133
+ "name": name, "done": True, "media_entries": [entry],
1134
+ "status": m_status, "error": None,
1135
+ })
1136
+ return ops_summary, raw_polls
1137
+
1138
+ async def _fetch_workflow_bytes(self, mid: str) -> Optional[dict[str, Any]]:
1139
+ """Fetch a finished workflow clip's encoded MP4 via ``GET /v1/media/<id>``.
1140
+
1141
+ Returns a media entry (``{media_id, url, mediaType, encoded_video}``)
1142
+ once Flow serves the base64 MP4, or ``None`` while the bytes are not
1143
+ yet available (any error / non-``ftyp`` payload) so the caller keeps
1144
+ polling.
1145
+ """
1146
+ import base64 as _b64
1147
+
1148
+ try:
1149
+ resp = await self._client.api_request(
1150
+ url=_media_get_url(mid),
1151
+ method="GET",
1152
+ headers=dict(_API_HEADERS),
1153
+ body=None,
1154
+ )
1155
+ except Exception as exc: # noqa: BLE001
1156
+ logger.warning("workflow byte fetch error for %s: %s", mid[:8], exc)
1157
+ return None
1158
+ if not isinstance(resp, dict):
1159
+ return None
1160
+ status_code = resp.get("status")
1161
+ if isinstance(status_code, int) and status_code >= 400:
1162
+ return None # not ready yet
1163
+ data = resp.get("data") if isinstance(resp.get("data"), dict) else {}
1164
+ video_block = data.get("video") if isinstance(data.get("video"), dict) else {}
1165
+ encoded = (
1166
+ video_block.get("encodedVideo") if isinstance(video_block, dict) else None
1167
+ )
1168
+ if not isinstance(encoded, str) or not encoded:
1169
+ return None
1170
+ try:
1171
+ binary = _b64.b64decode(encoded, validate=False)
1172
+ except Exception: # noqa: BLE001
1173
+ return None
1174
+ # MP4 box layout: bytes 4..8 == "ftyp" on a complete file.
1175
+ if not (len(binary) >= 12 and binary[4:8] == b"ftyp"):
1176
+ return None
1177
+ fife = (
1178
+ video_block.get("fifeUrl") if isinstance(video_block, dict) else None
1179
+ ) or data.get("fifeUrl")
1180
+ return {
1181
+ "media_id": mid,
1182
+ "url": fife if isinstance(fife, str) else None,
1183
+ "mediaType": "video",
1184
+ "encoded_video": encoded,
1185
+ }
1186
+
1187
+ # ── image generation (api_request + captcha) ───────────────────────────
1188
+ async def gen_image(
1189
+ self,
1190
+ prompt: str,
1191
+ project_id: str,
1192
+ aspect_ratio: str = "IMAGE_ASPECT_RATIO_LANDSCAPE",
1193
+ paygate_tier: Optional[str] = None,
1194
+ ref_media_ids: Optional[list[str]] = None,
1195
+ variant_count: int = 1,
1196
+ character_media_ids: Optional[list[str]] = None, # legacy alias
1197
+ prompts: Optional[list[str]] = None,
1198
+ image_model: Optional[str] = None,
1199
+ ) -> dict[str, Any]:
1200
+ """Generate ``variant_count`` images (1-4). When ``ref_media_ids`` is
1201
+ provided, every request item is augmented with ``imageInputs`` so Flow
1202
+ conditions the result on those upstream images (any combination of
1203
+ character / image / visual_asset upstream nodes — all become
1204
+ ``IMAGE_INPUT_TYPE_REFERENCE`` inputs).
1205
+
1206
+ Multiple variants are produced by replicating the request item with
1207
+ distinct seeds — Flow returns one entry in ``data.media[]`` per
1208
+ request item.
1209
+
1210
+ ``paygate_tier`` is required. See ``gen_video`` for rationale.
1211
+ """
1212
+ if paygate_tier is None:
1213
+ raise ValueError("paygate_tier is required — caller must resolve before dispatch")
1214
+ n = max(1, min(int(variant_count), MAX_VARIANT_COUNT))
1215
+ ts = int(time.time() * 1000)
1216
+ ctx = _client_context(project_id, paygate_tier)
1217
+ model_name = resolve_image_model(image_model)
1218
+ # Accept the legacy `character_media_ids` kwarg as a fallback.
1219
+ merged_refs = ref_media_ids if ref_media_ids is not None else character_media_ids
1220
+ image_inputs = None
1221
+ if merged_refs:
1222
+ image_inputs = [
1223
+ {"name": mid, "imageInputType": "IMAGE_INPUT_TYPE_REFERENCE"}
1224
+ for mid in merged_refs
1225
+ ]
1226
+
1227
+ # Per-variant prompts: when the caller provides `prompts`, each
1228
+ # request_item gets its own text so the 4 variants render with
1229
+ # different poses instead of 4 seeds of one stance. Missing /
1230
+ # short list falls back to the single `prompt` for that slot.
1231
+ per_item_prompts: list[str] = []
1232
+ for i in range(n):
1233
+ if prompts and i < len(prompts) and isinstance(prompts[i], str) and prompts[i]:
1234
+ per_item_prompts.append(prompts[i])
1235
+ else:
1236
+ per_item_prompts.append(prompt)
1237
+
1238
+ requests_arr: list[dict[str, Any]] = []
1239
+ for i in range(n):
1240
+ seed = (ts + i * 9973) % 1_000_000 # any deterministic spread is fine
1241
+ item: dict[str, Any] = {
1242
+ "clientContext": {**ctx, "sessionId": f";{ts + i}"},
1243
+ "seed": seed,
1244
+ "structuredPrompt": {"parts": [{"text": per_item_prompts[i]}]},
1245
+ "imageAspectRatio": aspect_ratio,
1246
+ "imageModelName": model_name,
1247
+ }
1248
+ if image_inputs is not None:
1249
+ item["imageInputs"] = list(image_inputs)
1250
+ requests_arr.append(item)
1251
+
1252
+ body = {
1253
+ "clientContext": ctx,
1254
+ "mediaGenerationContext": {"batchId": str(uuid.uuid4())},
1255
+ "useNewMedia": True,
1256
+ "requests": requests_arr,
1257
+ }
1258
+
1259
+ resp = await self._client.api_request(
1260
+ url=_generate_images_url(project_id),
1261
+ method="POST",
1262
+ headers=dict(_API_HEADERS),
1263
+ body=body,
1264
+ captcha_action=CAPTCHA_IMAGE,
1265
+ )
1266
+ if isinstance(resp, dict) and resp.get("error"):
1267
+ return {"raw": resp, "error": resp["error"]}
1268
+ inner_err = _extract_inner_api_error(resp)
1269
+ if inner_err:
1270
+ return {"raw": resp, "error": inner_err}
1271
+
1272
+ entries = extract_media_entries(resp)
1273
+ media_ids = [e["media_id"] for e in entries]
1274
+ return {"raw": resp, "media_ids": media_ids, "media_entries": entries}
1275
+
1276
+ # ── image refine (edit_image) ──────────────────────────────────────────
1277
+ async def edit_image(
1278
+ self,
1279
+ prompt: str,
1280
+ project_id: str,
1281
+ source_media_id: str,
1282
+ ref_media_ids: Optional[list[str]] = None,
1283
+ aspect_ratio: str = "IMAGE_ASPECT_RATIO_LANDSCAPE",
1284
+ paygate_tier: Optional[str] = None,
1285
+ image_model: Optional[str] = None,
1286
+ ) -> dict[str, Any]:
1287
+ """Refine an existing image with an optional list of reference media.
1288
+
1289
+ Order of ``imageInputs`` matters — flowkit puts BASE_IMAGE first so
1290
+ Flow knows which is the canonical source.
1291
+
1292
+ ``paygate_tier`` is required. See ``gen_video`` for rationale.
1293
+ """
1294
+ if paygate_tier is None:
1295
+ raise ValueError("paygate_tier is required — caller must resolve before dispatch")
1296
+ ts = int(time.time() * 1000)
1297
+ ctx = _client_context(project_id, paygate_tier)
1298
+ model_name = resolve_image_model(image_model)
1299
+
1300
+ image_inputs: list[dict[str, Any]] = [
1301
+ {"name": source_media_id, "imageInputType": "IMAGE_INPUT_TYPE_BASE_IMAGE"}
1302
+ ]
1303
+ for mid in ref_media_ids or []:
1304
+ if isinstance(mid, str) and mid:
1305
+ image_inputs.append(
1306
+ {"name": mid, "imageInputType": "IMAGE_INPUT_TYPE_REFERENCE"}
1307
+ )
1308
+
1309
+ request_item = {
1310
+ "clientContext": {**ctx, "sessionId": f";{ts}"},
1311
+ "seed": ts % 1_000_000,
1312
+ "structuredPrompt": {"parts": [{"text": prompt}]},
1313
+ "imageAspectRatio": aspect_ratio,
1314
+ "imageModelName": model_name,
1315
+ "imageInputs": image_inputs,
1316
+ }
1317
+ body = {
1318
+ "clientContext": ctx,
1319
+ "mediaGenerationContext": {"batchId": str(uuid.uuid4())},
1320
+ "useNewMedia": True,
1321
+ "requests": [request_item],
1322
+ }
1323
+
1324
+ resp = await self._client.api_request(
1325
+ url=_generate_images_url(project_id),
1326
+ method="POST",
1327
+ headers=dict(_API_HEADERS),
1328
+ body=body,
1329
+ captcha_action=CAPTCHA_IMAGE,
1330
+ )
1331
+ if isinstance(resp, dict) and resp.get("error"):
1332
+ return {"raw": resp, "error": resp["error"]}
1333
+ inner_err = _extract_inner_api_error(resp)
1334
+ if inner_err:
1335
+ return {"raw": resp, "error": inner_err}
1336
+
1337
+ entries = extract_media_entries(resp)
1338
+ media_ids = [e["media_id"] for e in entries]
1339
+ return {"raw": resp, "media_ids": media_ids, "media_entries": entries}
1340
+
1341
+ # ── image upload (api_request, no captcha) ─────────────────────────────
1342
+ async def upload_image(
1343
+ self,
1344
+ image_base64: str,
1345
+ mime_type: str,
1346
+ project_id: str,
1347
+ file_name: str = "upload.png",
1348
+ ) -> dict[str, Any]:
1349
+ """Upload a user-provided image into a Flow project. Returns
1350
+ ``{raw, media_id}`` on success or ``{raw, error}`` on failure.
1351
+
1352
+ ``image_base64`` should be a base64-encoded payload (no data: prefix).
1353
+ Flow accepts the image bytes inline in the JSON body.
1354
+ """
1355
+ body = {
1356
+ "clientContext": {
1357
+ "projectId": str(project_id),
1358
+ "tool": "PINHOLE",
1359
+ },
1360
+ "fileName": file_name,
1361
+ "imageBytes": image_base64,
1362
+ "isHidden": False,
1363
+ "isUserUploaded": True,
1364
+ "mimeType": mime_type,
1365
+ }
1366
+ resp = await self._client.api_request(
1367
+ url=UPLOAD_IMAGE_URL,
1368
+ method="POST",
1369
+ headers=dict(_API_HEADERS),
1370
+ body=body,
1371
+ )
1372
+ if isinstance(resp, dict) and resp.get("error"):
1373
+ return {"raw": resp, "error": resp["error"]}
1374
+
1375
+ media_id = _extract_uploaded_media_id(resp)
1376
+ if media_id is None:
1377
+ # Flow returned 200 but no usable media handle. Most common cause
1378
+ # is a silent content-filter rejection (logos/watermarks/branded
1379
+ # imagery from product CDNs); next most common is a Flow schema
1380
+ # change. Log the full payload so the operator can tell which.
1381
+ logger.error(
1382
+ "upload_image: no media_id in response (project_id=%s, "
1383
+ "file=%s, mime=%s) — raw=%r",
1384
+ project_id, file_name, mime_type, resp,
1385
+ )
1386
+ return {"raw": resp, "error": "no_media_id_in_upload_response"}
1387
+ return {"raw": resp, "media_id": media_id}
1388
+
1389
+ # ── video upload (resumable, via extension) ────────────────────────────
1390
+ async def upload_video(
1391
+ self,
1392
+ video_base64: str,
1393
+ mime_type: str,
1394
+ project_id: str,
1395
+ file_name: str = "upload.mp4",
1396
+ end_offset: str = "8s",
1397
+ ) -> dict[str, Any]:
1398
+ """Upload a user-provided video into a Flow project, returning
1399
+ ``{raw, media_id, workflow_id, width, height}`` on success or
1400
+ ``{raw, error}`` on failure.
1401
+
1402
+ Unlike images, video upload can't go through ``api_request`` — it's a
1403
+ resumable, cookie-authed handshake against labs.google with a binary
1404
+ PUT body. The extension runs that handshake in the browser session (a
1405
+ ``upload_video`` bridge method) and hands back Flow's final
1406
+ ``{mediaServerId, workflowServerId, videoWidth, videoHeight}`` payload.
1407
+ We then register the trim window via ``videoFx.updateVideoOffset`` so
1408
+ the upload becomes a referenceable workflow media — exactly what the
1409
+ Flow web client does after a drag-and-drop upload.
1410
+
1411
+ ``video_base64`` should be a base64-encoded payload (no data: prefix).
1412
+ """
1413
+ resp = await self._client.upload_video(
1414
+ project_id=project_id,
1415
+ file_name=file_name,
1416
+ content_type=mime_type,
1417
+ data_base64=video_base64,
1418
+ )
1419
+ if isinstance(resp, dict) and resp.get("error"):
1420
+ return {"raw": resp, "error": resp["error"]}
1421
+
1422
+ final = resp.get("data") if isinstance(resp, dict) else None
1423
+ final = final if isinstance(final, dict) else {}
1424
+ media_id = final.get("mediaServerId")
1425
+ if not isinstance(media_id, str) or not media_id:
1426
+ logger.error(
1427
+ "upload_video: no mediaServerId in response (project_id=%s, "
1428
+ "file=%s, mime=%s) — raw=%r",
1429
+ project_id, file_name, mime_type, resp,
1430
+ )
1431
+ return {"raw": resp, "error": "no_media_id_in_upload_response"}
1432
+
1433
+ # Register the trim window (videoFx.updateVideoOffset). This is what
1434
+ # promotes the raw upload into a usable workflow media; the media id is
1435
+ # already valid, so a failure here is soft — surface it in raw for
1436
+ # diagnosis but still return the id.
1437
+ offset_resp = await self._client.trpc_request(
1438
+ url=TRPC_UPDATE_VIDEO_OFFSET,
1439
+ method="POST",
1440
+ headers=_TRPC_HEADERS,
1441
+ body={
1442
+ "json": {
1443
+ "mediaId": media_id,
1444
+ "startOffset": "0s",
1445
+ "endOffset": end_offset,
1446
+ }
1447
+ },
1448
+ )
1449
+ workflow_id = final.get("workflowServerId")
1450
+ return {
1451
+ "raw": {"upload": resp, "offset": offset_resp},
1452
+ "media_id": media_id,
1453
+ "workflow_id": workflow_id if isinstance(workflow_id, str) else None,
1454
+ "width": final.get("videoWidth"),
1455
+ "height": final.get("videoHeight"),
1456
+ }
1457
+
1458
+
1459
+ def _extract_project_id(resp: Any) -> Optional[str]:
1460
+ """TRPC createProject nests the projectId quite deeply."""
1461
+ try:
1462
+ data = resp.get("data") if isinstance(resp, dict) else None
1463
+ return data["result"]["data"]["json"]["result"]["projectId"] # type: ignore[index]
1464
+ except (KeyError, TypeError):
1465
+ return None
1466
+
1467
+
1468
+ _VALID_TIERS = {"PAYGATE_TIER_ONE", "PAYGATE_TIER_TWO"}
1469
+
1470
+
1471
+ def _extract_uploaded_media_id(resp: Any) -> Optional[str]:
1472
+ """uploadImage returns ``data.media.name`` as the new media_id."""
1473
+ if not isinstance(resp, dict):
1474
+ return None
1475
+ data = resp.get("data")
1476
+ if not isinstance(data, dict):
1477
+ return None
1478
+ media = data.get("media")
1479
+ if isinstance(media, dict):
1480
+ name = media.get("name")
1481
+ if isinstance(name, str) and name:
1482
+ return name
1483
+ return None
1484
+
1485
+
1486
+ def extract_operation_names(resp: Any) -> list[str]:
1487
+ """Pull ``operation.name`` out of a ``batchAsyncGenerateVideo*`` response.
1488
+
1489
+ Supports two shapes:
1490
+
1491
+ * **OLD** (Lite / Fast / Quality) — ``data.operations[].operation.name``.
1492
+ * **NEW** (Low Priority — ``_low_priority`` / ``_relaxed`` models) —
1493
+ ``data.workflows[].name``. Workflows don't have ``operation.name``;
1494
+ callers that need to poll must also read ``primaryMediaId`` from
1495
+ ``workflows[].metadata`` (see ``extract_video_workflows``).
1496
+ """
1497
+ if not isinstance(resp, dict):
1498
+ return []
1499
+ data = resp.get("data")
1500
+ if not isinstance(data, dict):
1501
+ return []
1502
+ names: list[str] = []
1503
+ ops = data.get("operations")
1504
+ if isinstance(ops, list):
1505
+ for op in ops:
1506
+ if not isinstance(op, dict):
1507
+ continue
1508
+ inner = op.get("operation") if isinstance(op.get("operation"), dict) else None
1509
+ if inner is None:
1510
+ # Some variants inline the name at top level.
1511
+ name = op.get("name")
1512
+ else:
1513
+ name = inner.get("name")
1514
+ if isinstance(name, str) and name:
1515
+ names.append(name)
1516
+ if names:
1517
+ return names
1518
+ # NEW workflow schema. The tracking key is the workflow id — take it from
1519
+ # `media[].workflowId` (preferred, pairs with extract_video_workflows) or
1520
+ # fall back to `workflows[].name`. The two normally coincide; keeping them
1521
+ # consistent ensures check_async routes these to workflow polling, not the
1522
+ # operation poll (which would 400 on a workflow id).
1523
+ media = data.get("media")
1524
+ if isinstance(media, list):
1525
+ for m in media:
1526
+ wf = m.get("workflowId") if isinstance(m, dict) else None
1527
+ if isinstance(wf, str) and wf and wf not in names:
1528
+ names.append(wf)
1529
+ if names:
1530
+ return names
1531
+ workflows = data.get("workflows")
1532
+ if isinstance(workflows, list):
1533
+ for wf in workflows:
1534
+ if not isinstance(wf, dict):
1535
+ continue
1536
+ name = wf.get("name")
1537
+ if isinstance(name, str) and name:
1538
+ names.append(name)
1539
+ return names
1540
+
1541
+
1542
+ def extract_video_workflows(resp: Any) -> list[dict[str, Any]]:
1543
+ """Pull workflow entries out of a NEW-schema video submit response.
1544
+
1545
+ Returns ``[{"name": <workflow_name>, "primary_media_id": <new media uuid>},
1546
+ ...]`` — the media id to poll for the freshly generated output. Empty list
1547
+ when the response is OLD-schema (operations-based) or has no workflows.
1548
+
1549
+ Source of the media id matters: ``data.media[].name`` is the NEW output, and
1550
+ we pair it to its workflow via ``media[].workflowId``. For **edit** the
1551
+ workflow's ``metadata.primaryMediaId`` still points at the SOURCE clip, so
1552
+ polling that would return the un-edited input — we must poll ``media[].name``
1553
+ instead. For t2v/r2v/extend the two coincide, so ``media[]`` is correct in
1554
+ all cases. ``workflows[].metadata.primaryMediaId`` is only a fallback for a
1555
+ response that omits ``media[]``.
1556
+ """
1557
+ if not isinstance(resp, dict):
1558
+ return []
1559
+ data = resp.get("data")
1560
+ if not isinstance(data, dict):
1561
+ return []
1562
+
1563
+ out: list[dict[str, Any]] = []
1564
+ media = data.get("media")
1565
+ if isinstance(media, list):
1566
+ for m in media:
1567
+ if not isinstance(m, dict):
1568
+ continue
1569
+ mid = m.get("name")
1570
+ wf = m.get("workflowId")
1571
+ if isinstance(mid, str) and mid and isinstance(wf, str) and wf:
1572
+ out.append({"name": wf, "primary_media_id": mid})
1573
+ if out:
1574
+ return out
1575
+
1576
+ # Fallback: derive from workflows[].metadata.primaryMediaId (no media[]).
1577
+ workflows = data.get("workflows")
1578
+ if not isinstance(workflows, list):
1579
+ return []
1580
+ for wf in workflows:
1581
+ if not isinstance(wf, dict):
1582
+ continue
1583
+ name = wf.get("name")
1584
+ meta = wf.get("metadata") if isinstance(wf.get("metadata"), dict) else {}
1585
+ primary = meta.get("primaryMediaId") if isinstance(meta, dict) else None
1586
+ if isinstance(name, str) and name and isinstance(primary, str) and primary:
1587
+ out.append({"name": name, "primary_media_id": primary})
1588
+ return out
1589
+
1590
+
1591
+ def extract_video_operations(
1592
+ resp: Any, *, requested: list[str]
1593
+ ) -> list[dict[str, Any]]:
1594
+ """Summarise a ``batchCheckAsync`` response.
1595
+
1596
+ Flow's response shape is::
1597
+
1598
+ {"data": {"operations": [{
1599
+ "status": "MEDIA_GENERATION_STATUS_{PENDING,SUCCESSFUL,FAILED}",
1600
+ "operation": {"name": "<id>", "metadata": {"video": {
1601
+ "mediaId": "<uuid>", "fifeUrl": "https://flow-content..."
1602
+ }}}
1603
+ }]}}
1604
+
1605
+ flowkit treats ``MEDIA_GENERATION_STATUS_SUCCESSFUL`` as terminal-success;
1606
+ we mirror that.
1607
+
1608
+ Returns one entry per *requested* operation name, in order. Missing
1609
+ operations are reported as ``done=False`` so the caller can keep polling.
1610
+ """
1611
+ by_name: dict[str, dict[str, Any]] = {}
1612
+ if isinstance(resp, dict):
1613
+ data = resp.get("data")
1614
+ if isinstance(data, dict):
1615
+ ops = data.get("operations")
1616
+ if isinstance(ops, list):
1617
+ for op in ops:
1618
+ if not isinstance(op, dict):
1619
+ continue
1620
+ inner = op.get("operation") if isinstance(op.get("operation"), dict) else op
1621
+ name = inner.get("name") if isinstance(inner, dict) else None
1622
+ if not isinstance(name, str):
1623
+ continue
1624
+ meta = (inner.get("metadata") or {}) if isinstance(inner, dict) else {}
1625
+ video_meta = meta.get("video") if isinstance(meta.get("video"), dict) else {}
1626
+ media_id = video_meta.get("mediaId") if isinstance(video_meta, dict) else None
1627
+ fife = video_meta.get("fifeUrl") if isinstance(video_meta, dict) else None
1628
+ # Flow's video poll response usually omits `mediaId` and only
1629
+ # provides `mediaGenerationId` (base64 protobuf, NOT a UUID).
1630
+ # The actual UUID is embedded in the `fifeUrl` path. Recover it.
1631
+ if not (isinstance(media_id, str) and media_id):
1632
+ recovered = _media_id_from_url(fife if isinstance(fife, str) else None)
1633
+ if recovered is None and isinstance(video_meta, dict):
1634
+ recovered = _media_id_from_url(video_meta.get("servingBaseUri"))
1635
+ if recovered is not None:
1636
+ media_id = recovered
1637
+ # Flow puts the status at the *top* of each op envelope,
1638
+ # not on the inner operation object — bug we hit before.
1639
+ status = op.get("status") if isinstance(op.get("status"), str) else None
1640
+ # Per-op terminal failure (e.g. PUBLIC_ERROR_AUDIO_FILTERED).
1641
+ # Flow puts the error on the inner operation object as
1642
+ # ``{code, message}``. We surface it so the worker can bail
1643
+ # instead of polling for the full timeout.
1644
+ op_err: Optional[str] = None
1645
+ inner_err = inner.get("error") if isinstance(inner, dict) else None
1646
+ if isinstance(inner_err, dict):
1647
+ msg = inner_err.get("message") or inner_err.get("status") or "operation_failed"
1648
+ op_err = str(msg)
1649
+ if status == "MEDIA_GENERATION_STATUS_FAILED" and op_err is None:
1650
+ op_err = "MEDIA_GENERATION_STATUS_FAILED"
1651
+ done_flag = (
1652
+ status == "MEDIA_GENERATION_STATUS_SUCCESSFUL"
1653
+ or status == "MEDIA_GENERATION_STATUS_FAILED"
1654
+ or bool(inner.get("done"))
1655
+ or bool(media_id and fife)
1656
+ )
1657
+ entries = []
1658
+ if (
1659
+ done_flag
1660
+ and op_err is None
1661
+ and isinstance(media_id, str)
1662
+ ):
1663
+ entries.append(
1664
+ {
1665
+ "media_id": media_id,
1666
+ "url": fife if isinstance(fife, str) else None,
1667
+ "mediaType": "video",
1668
+ }
1669
+ )
1670
+ by_name[name] = {
1671
+ "name": name,
1672
+ "done": done_flag,
1673
+ "media_entries": entries,
1674
+ "status": status,
1675
+ "error": op_err,
1676
+ }
1677
+
1678
+ out: list[dict[str, Any]] = []
1679
+ for name in requested:
1680
+ out.append(
1681
+ by_name.get(
1682
+ name, {"name": name, "done": False, "media_entries": []}
1683
+ )
1684
+ )
1685
+ return out
1686
+
1687
+
1688
+ def _extract_media_ids(resp: Any) -> list[str]:
1689
+ return [e["media_id"] for e in extract_media_entries(resp)]
1690
+
1691
+
1692
+ def extract_media_entries(resp: Any) -> list[dict[str, Any]]:
1693
+ """Pull media entries out of a ``batchGenerateImages`` response.
1694
+
1695
+ Returns a list of ``{media_id, url, mediaType}`` dicts suitable for
1696
+ ``media.ingest_urls``. ``url`` may be missing if Flow didn't include a
1697
+ ``fifeUrl`` for some reason — caller should handle that.
1698
+ """
1699
+ if not isinstance(resp, dict):
1700
+ return []
1701
+ data = resp.get("data")
1702
+ if not isinstance(data, dict):
1703
+ return []
1704
+ media = data.get("media")
1705
+ if not isinstance(media, list):
1706
+ return []
1707
+ out: list[dict[str, Any]] = []
1708
+ for m in media:
1709
+ if not isinstance(m, dict):
1710
+ continue
1711
+ media_id = m.get("name")
1712
+ if not isinstance(media_id, str) or not media_id:
1713
+ continue
1714
+ url: Optional[str] = None
1715
+ kind = "image"
1716
+ image = m.get("image") if isinstance(m.get("image"), dict) else None
1717
+ video = m.get("video") if isinstance(m.get("video"), dict) else None
1718
+ if image is not None:
1719
+ gen = image.get("generatedImage")
1720
+ if isinstance(gen, dict):
1721
+ candidate = gen.get("fifeUrl")
1722
+ if isinstance(candidate, str):
1723
+ url = candidate
1724
+ kind = "image"
1725
+ elif video is not None:
1726
+ gen = video.get("generatedVideo") or video.get("generatedImage")
1727
+ if isinstance(gen, dict):
1728
+ candidate = gen.get("fifeUrl")
1729
+ if isinstance(candidate, str):
1730
+ url = candidate
1731
+ kind = "video"
1732
+ out.append({"media_id": media_id, "url": url, "mediaType": kind})
1733
+ return out
1734
+
1735
+
1736
+ _sdk: Optional[FlowSDK] = None
1737
+
1738
+
1739
+ def get_flow_sdk() -> FlowSDK:
1740
+ global _sdk
1741
+ if _sdk is None:
1742
+ _sdk = FlowSDK()
1743
+ return _sdk