storyboard-bridge 0.7.1 → 0.7.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/index.mjs +1 -0
- package/package.json +1 -1
- package/tryinfer_client.py +96 -32
package/index.mjs
CHANGED
|
@@ -328,6 +328,7 @@ function handleTryinfer(msg, ws) {
|
|
|
328
328
|
'--aspect', String(p.aspectRatio || '1:1'),
|
|
329
329
|
'--resolution', String(p.resolution || '1080p'));
|
|
330
330
|
if (p.audio === false) args.push('--no-audio');
|
|
331
|
+
if (p.numImages && p.numImages > 1) args.push('--num', String(p.numImages)); // Seedream image batch (2x)
|
|
331
332
|
for (const u of (Array.isArray(p.imageUrls) ? p.imageUrls : [])) args.push('--image-url', String(u));
|
|
332
333
|
if (p.lastFrameUrl) args.push('--last-frame-url', String(p.lastFrameUrl)); // image-to-video END frame
|
|
333
334
|
|
package/package.json
CHANGED
package/tryinfer_client.py
CHANGED
|
@@ -123,29 +123,58 @@ def _fetch_expr(method, url, body):
|
|
|
123
123
|
return "".join(parts)
|
|
124
124
|
|
|
125
125
|
|
|
126
|
+
# CDP errors that mean the execution context we ran fetch() in is gone — the Studio tab NAVIGATED (it changes
|
|
127
|
+
# route when a generation starts/finishes) or reloaded. We recover by re-attaching to the tab's fresh context.
|
|
128
|
+
_CTX_DEAD = ("navigated", "context", "closed", "-32000", "detached", "Session with given id")
|
|
129
|
+
|
|
130
|
+
|
|
126
131
|
class BrowserSession:
|
|
127
132
|
def __init__(self, cdp_http, match):
|
|
133
|
+
self.cdp_http = cdp_http
|
|
134
|
+
self.match = match
|
|
128
135
|
ver = http_json(f"{cdp_http}/json/version")
|
|
129
136
|
log(f"Connected to {ver.get('Browser')}")
|
|
137
|
+
self.browser_ws = ver["webSocketDebuggerUrl"]
|
|
138
|
+
self.cdp = None
|
|
139
|
+
self._attach()
|
|
140
|
+
|
|
141
|
+
def _attach(self):
|
|
142
|
+
"""(Re)resolve the matching tab and attach a fresh Runtime session. Safe to call repeatedly — used
|
|
143
|
+
both at startup and to recover after the Studio tab navigates mid-poll (which kills the context)."""
|
|
144
|
+
if self.cdp:
|
|
145
|
+
try: self.cdp.close()
|
|
146
|
+
except Exception: pass
|
|
130
147
|
tid = url = None
|
|
131
|
-
for t in http_json(f"{cdp_http}/json/list"):
|
|
132
|
-
if t.get("type") == "page" and match in (t.get("url") or ""):
|
|
148
|
+
for t in http_json(f"{self.cdp_http}/json/list"):
|
|
149
|
+
if t.get("type") == "page" and self.match in (t.get("url") or ""):
|
|
133
150
|
tid, url = t["id"], t["url"]
|
|
134
151
|
break
|
|
135
152
|
if not tid:
|
|
136
|
-
raise RuntimeError(f"No open tab whose URL contains '{match}'. Open tryinfer.com and log in.")
|
|
137
|
-
|
|
138
|
-
self.cdp = CDP(ver["webSocketDebuggerUrl"])
|
|
153
|
+
raise RuntimeError(f"No open tab whose URL contains '{self.match}'. Open tryinfer.com and log in.")
|
|
154
|
+
self.cdp = CDP(self.browser_ws)
|
|
139
155
|
self.session_id = self.cdp.call("Target.attachToTarget", {"targetId": tid, "flatten": True})["sessionId"]
|
|
140
156
|
self.cdp.call("Runtime.enable", session_id=self.session_id)
|
|
157
|
+
log(f"Routing API calls through tab: {url}")
|
|
141
158
|
|
|
142
|
-
def request(self, method, url, body=None, timeout=120, retries=
|
|
159
|
+
def request(self, method, url, body=None, timeout=120, retries=4):
|
|
143
160
|
last = None
|
|
144
161
|
for attempt in range(retries):
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
162
|
+
try:
|
|
163
|
+
res = self.cdp.call(
|
|
164
|
+
"Runtime.evaluate",
|
|
165
|
+
{"expression": _fetch_expr(method, url, body), "awaitPromise": True, "returnByValue": True},
|
|
166
|
+
session_id=self.session_id, timeout=timeout)
|
|
167
|
+
except Exception as e:
|
|
168
|
+
last = f"CDP evaluate error: {e}"
|
|
169
|
+
# tab navigated / context died → re-attach to the new context and retry (don't lose the poll).
|
|
170
|
+
if any(s in str(e) for s in _CTX_DEAD):
|
|
171
|
+
log(f" tab navigated — re-attaching… ({e})")
|
|
172
|
+
time.sleep(1)
|
|
173
|
+
try: self._attach()
|
|
174
|
+
except Exception as e2: last = f"re-attach failed: {e2}"
|
|
175
|
+
if attempt < retries - 1:
|
|
176
|
+
time.sleep(2); continue
|
|
177
|
+
raise RuntimeError(last)
|
|
149
178
|
val = res.get("result", {}).get("value")
|
|
150
179
|
if val is None:
|
|
151
180
|
last = f"evaluate failed: {res.get('exceptionDetails')}"
|
|
@@ -175,12 +204,19 @@ FAILED = {"FAILED", "ERROR", "CANCELLED", "CANCELED"}
|
|
|
175
204
|
|
|
176
205
|
|
|
177
206
|
def submit(session, prompt, image_urls, duration, aspect, resolution, audio,
|
|
178
|
-
capability="reference-to-video", model="seedance-2.0-pro", last_frame_url=None
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
•
|
|
182
|
-
|
|
183
|
-
|
|
207
|
+
capability="reference-to-video", model="seedance-2.0-pro", last_frame_url=None,
|
|
208
|
+
num_images=1, input_json=None):
|
|
209
|
+
"""POST a generation task. Video + image shapes, all URL-based (no upload). Returns the task id.
|
|
210
|
+
• reference-to-video : reference_image_urls[] + resolution + audio
|
|
211
|
+
• image-to-video : image_url (start) [+ last_frame_image_url] + audio (no resolution)
|
|
212
|
+
• edit (Seedream) : image_url + prompt + num_images (UI hides aspect)
|
|
213
|
+
• text-to-image : prompt + num_images + aspect_ratio (no image)
|
|
214
|
+
`input_json` overrides the WHOLE input dict verbatim — for probing undocumented combos."""
|
|
215
|
+
medium = "image" if capability in ("edit", "text-to-image") else "video"
|
|
216
|
+
if input_json is not None:
|
|
217
|
+
inp = json.loads(input_json) # PROBE: send exactly this
|
|
218
|
+
meta = {"prompt": prompt, "medium": medium}
|
|
219
|
+
elif capability == "image-to-video":
|
|
184
220
|
if not image_urls:
|
|
185
221
|
raise RuntimeError("image-to-video needs a start frame (--image-url)")
|
|
186
222
|
inp = {"image_url": image_urls[0], "prompt": prompt, "duration_seconds": duration, "aspect_ratio": aspect, "audio": audio}
|
|
@@ -189,6 +225,14 @@ def submit(session, prompt, image_urls, duration, aspect, resolution, audio,
|
|
|
189
225
|
meta = {"prompt": prompt, "medium": "video", "ratio": RATIO_WORD.get(aspect, "square"), "kind": "animate", "sourceUrl": image_urls[0]}
|
|
190
226
|
if last_frame_url:
|
|
191
227
|
meta["lastFrameUrl"] = last_frame_url
|
|
228
|
+
elif capability == "edit":
|
|
229
|
+
if not image_urls:
|
|
230
|
+
raise RuntimeError("edit needs a source image (--image-url)")
|
|
231
|
+
inp = {"image_url": image_urls[0], "prompt": prompt, "num_images": num_images}
|
|
232
|
+
meta = {"prompt": prompt, "medium": "image", "kind": "edit", "sourceUrl": image_urls[0]}
|
|
233
|
+
elif capability == "text-to-image":
|
|
234
|
+
inp = {"prompt": prompt, "num_images": num_images, "aspect_ratio": aspect}
|
|
235
|
+
meta = {"prompt": prompt, "medium": "image", "kind": "generate"}
|
|
192
236
|
else: # reference-to-video
|
|
193
237
|
inp = {"reference_image_urls": image_urls, "prompt": prompt, "duration_seconds": duration, "aspect_ratio": aspect, "resolution": resolution, "audio": audio}
|
|
194
238
|
meta = {"prompt": prompt, "medium": "video", "ratio": RATIO_WORD.get(aspect, "square"), "kind": "reference", "referenceImageUrls": image_urls, "referenceRequestIds": []}
|
|
@@ -224,26 +268,33 @@ def poll(session, task_id, on_status=None, timeout=1800, interval=3):
|
|
|
224
268
|
|
|
225
269
|
|
|
226
270
|
def get_result(session, task_id):
|
|
227
|
-
"""GET the result
|
|
271
|
+
"""GET the result. Returns {'videoUrl':…} for video, or {'images':[url,…]} for image. Raises on
|
|
272
|
+
moderation block / empty output."""
|
|
228
273
|
j = session.request("GET", f"{API}/create/generation-tasks/{task_id}/result").json()
|
|
229
274
|
mod = j.get("moderation_status")
|
|
230
275
|
if mod and mod != "allowed":
|
|
231
276
|
raise RuntimeError(f"moderation blocked: {mod}")
|
|
232
277
|
out = j.get("output") or {}
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
278
|
+
if out.get("video_url"):
|
|
279
|
+
return {"videoUrl": out["video_url"], "output": out}
|
|
280
|
+
imgs = [x.get("url") for x in (out.get("images") or []) if isinstance(x, dict) and x.get("url")]
|
|
281
|
+
if imgs:
|
|
282
|
+
return {"images": imgs, "output": out}
|
|
283
|
+
raise RuntimeError(f"no video_url/images in result: {json.dumps(j)[:800]}")
|
|
237
284
|
|
|
238
285
|
|
|
239
286
|
def main():
|
|
240
287
|
ap = argparse.ArgumentParser()
|
|
241
288
|
ap.add_argument("--prompt", default="")
|
|
242
|
-
ap.add_argument("--capability", default="reference-to-video",
|
|
243
|
-
|
|
289
|
+
ap.add_argument("--capability", default="reference-to-video",
|
|
290
|
+
choices=["reference-to-video", "image-to-video", "edit", "text-to-image"])
|
|
291
|
+
ap.add_argument("--model", default=None, help="defaults by capability: image→seedream-5.0-pro, video→seedance-2.0-pro")
|
|
244
292
|
ap.add_argument("--image-url", action="append", default=[],
|
|
245
|
-
help="public image URL. reference-to-video: repeat for ordered refs. image-to-video: first =
|
|
293
|
+
help="public image URL. reference-to-video: repeat for ordered refs. image-to-video/edit: first = start/source.")
|
|
246
294
|
ap.add_argument("--last-frame-url", default=None, help="image-to-video only: END frame URL (optional)")
|
|
295
|
+
ap.add_argument("--num", type=int, default=1, help="num_images (image capabilities) — probe >1 here")
|
|
296
|
+
ap.add_argument("--input-json", default=None,
|
|
297
|
+
help="PROBE: raw JSON for args.input verbatim (override the built payload — test undocumented combos)")
|
|
247
298
|
ap.add_argument("--duration", type=int, default=5)
|
|
248
299
|
ap.add_argument("--aspect", default="1:1")
|
|
249
300
|
ap.add_argument("--resolution", default="1080p")
|
|
@@ -260,6 +311,8 @@ def main():
|
|
|
260
311
|
_JSON_OUT = sys.stdout
|
|
261
312
|
sys.stdout = sys.stderr # any stray print() can't corrupt the event stream
|
|
262
313
|
|
|
314
|
+
# model default is capability-aware: image caps use seedream (seedANCE is the VIDEO model — a common mixup).
|
|
315
|
+
model = args.model or ("seedream-5.0-pro" if args.capability in ("edit", "text-to-image") else "seedance-2.0-pro")
|
|
263
316
|
image_urls = args.image_url or ["https://www.gstatic.com/webp/gallery/1.jpg"] # default = permissive test image
|
|
264
317
|
sess = BrowserSession(f"http://{args.host}:{args.port}", args.match)
|
|
265
318
|
try:
|
|
@@ -269,11 +322,12 @@ def main():
|
|
|
269
322
|
event(event="progress", phase="rendering", taskId=task_id)
|
|
270
323
|
else:
|
|
271
324
|
task_id = submit(sess, args.prompt, image_urls, args.duration, args.aspect, args.resolution, not args.no_audio,
|
|
272
|
-
capability=args.capability, model=
|
|
325
|
+
capability=args.capability, model=model, last_frame_url=args.last_frame_url,
|
|
326
|
+
num_images=args.num, input_json=args.input_json)
|
|
273
327
|
log(f"submitted task {task_id}")
|
|
274
328
|
event(event="progress", phase="submitted", taskId=task_id)
|
|
275
329
|
poll(sess, task_id, on_status=lambda st: (log(f" status={st}"), event(event="progress", phase=st.lower(), taskId=task_id)))
|
|
276
|
-
|
|
330
|
+
res = get_result(sess, task_id)
|
|
277
331
|
except Exception as e:
|
|
278
332
|
event(event="error", reason="tryinfer", detail=str(e))
|
|
279
333
|
log(f"\n❌ {e}")
|
|
@@ -281,12 +335,22 @@ def main():
|
|
|
281
335
|
finally:
|
|
282
336
|
sess.close()
|
|
283
337
|
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
338
|
+
if res.get("images"):
|
|
339
|
+
event(event="done", taskId=task_id, images=res["images"])
|
|
340
|
+
dims = {x.get("url"): (x.get("width"), x.get("height")) for x in ((res.get("output") or {}).get("images") or []) if isinstance(x, dict)}
|
|
341
|
+
log(f"\n✅ {len(res['images'])} image(s):")
|
|
342
|
+
for u in res["images"]:
|
|
343
|
+
w, h = dims.get(u, (None, None))
|
|
344
|
+
log(f" {w}x{h} {u}")
|
|
345
|
+
if not args.emit_json:
|
|
346
|
+
print(json.dumps({"ok": True, "images": res["images"]}))
|
|
347
|
+
else:
|
|
348
|
+
out = res.get("output") or {}
|
|
349
|
+
event(event="done", taskId=task_id, videoUrl=res["videoUrl"],
|
|
350
|
+
width=out.get("width"), height=out.get("height"), duration=out.get("duration_seconds"))
|
|
351
|
+
log(f"\n✅ {out.get('width')}x{out.get('height')} {out.get('duration_seconds')}s")
|
|
352
|
+
if not args.emit_json:
|
|
353
|
+
print(json.dumps({"ok": True, "video_url": res["videoUrl"]}))
|
|
290
354
|
|
|
291
355
|
|
|
292
356
|
if __name__ == "__main__":
|