storyboard-bridge 0.7.1 → 0.7.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.mjs CHANGED
@@ -328,6 +328,7 @@ function handleTryinfer(msg, ws) {
328
328
  '--aspect', String(p.aspectRatio || '1:1'),
329
329
  '--resolution', String(p.resolution || '1080p'));
330
330
  if (p.audio === false) args.push('--no-audio');
331
+ if (p.numImages && p.numImages > 1) args.push('--num', String(p.numImages)); // Seedream image batch (2x)
331
332
  for (const u of (Array.isArray(p.imageUrls) ? p.imageUrls : [])) args.push('--image-url', String(u));
332
333
  if (p.lastFrameUrl) args.push('--last-frame-url', String(p.lastFrameUrl)); // image-to-video END frame
333
334
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "storyboard-bridge",
3
- "version": "0.7.1",
3
+ "version": "0.7.2",
4
4
  "description": "Desktop bridge that powers a hosted Storyboard AI webapp with your own local Claude Code / Gemini CLI login (no API key).",
5
5
  "type": "module",
6
6
  "bin": {
@@ -123,29 +123,58 @@ def _fetch_expr(method, url, body):
123
123
  return "".join(parts)
124
124
 
125
125
 
126
+ # CDP errors that mean the execution context we ran fetch() in is gone — the Studio tab NAVIGATED (it changes
127
+ # route when a generation starts/finishes) or reloaded. We recover by re-attaching to the tab's fresh context.
128
+ _CTX_DEAD = ("navigated", "context", "closed", "-32000", "detached", "Session with given id")
129
+
130
+
126
131
  class BrowserSession:
127
132
  def __init__(self, cdp_http, match):
133
+ self.cdp_http = cdp_http
134
+ self.match = match
128
135
  ver = http_json(f"{cdp_http}/json/version")
129
136
  log(f"Connected to {ver.get('Browser')}")
137
+ self.browser_ws = ver["webSocketDebuggerUrl"]
138
+ self.cdp = None
139
+ self._attach()
140
+
141
+ def _attach(self):
142
+ """(Re)resolve the matching tab and attach a fresh Runtime session. Safe to call repeatedly — used
143
+ both at startup and to recover after the Studio tab navigates mid-poll (which kills the context)."""
144
+ if self.cdp:
145
+ try: self.cdp.close()
146
+ except Exception: pass
130
147
  tid = url = None
131
- for t in http_json(f"{cdp_http}/json/list"):
132
- if t.get("type") == "page" and match in (t.get("url") or ""):
148
+ for t in http_json(f"{self.cdp_http}/json/list"):
149
+ if t.get("type") == "page" and self.match in (t.get("url") or ""):
133
150
  tid, url = t["id"], t["url"]
134
151
  break
135
152
  if not tid:
136
- raise RuntimeError(f"No open tab whose URL contains '{match}'. Open tryinfer.com and log in.")
137
- log(f"Routing API calls through tab: {url}")
138
- self.cdp = CDP(ver["webSocketDebuggerUrl"])
153
+ raise RuntimeError(f"No open tab whose URL contains '{self.match}'. Open tryinfer.com and log in.")
154
+ self.cdp = CDP(self.browser_ws)
139
155
  self.session_id = self.cdp.call("Target.attachToTarget", {"targetId": tid, "flatten": True})["sessionId"]
140
156
  self.cdp.call("Runtime.enable", session_id=self.session_id)
157
+ log(f"Routing API calls through tab: {url}")
141
158
 
142
- def request(self, method, url, body=None, timeout=120, retries=3):
159
+ def request(self, method, url, body=None, timeout=120, retries=4):
143
160
  last = None
144
161
  for attempt in range(retries):
145
- res = self.cdp.call(
146
- "Runtime.evaluate",
147
- {"expression": _fetch_expr(method, url, body), "awaitPromise": True, "returnByValue": True},
148
- session_id=self.session_id, timeout=timeout)
162
+ try:
163
+ res = self.cdp.call(
164
+ "Runtime.evaluate",
165
+ {"expression": _fetch_expr(method, url, body), "awaitPromise": True, "returnByValue": True},
166
+ session_id=self.session_id, timeout=timeout)
167
+ except Exception as e:
168
+ last = f"CDP evaluate error: {e}"
169
+ # tab navigated / context died → re-attach to the new context and retry (don't lose the poll).
170
+ if any(s in str(e) for s in _CTX_DEAD):
171
+ log(f" tab navigated — re-attaching… ({e})")
172
+ time.sleep(1)
173
+ try: self._attach()
174
+ except Exception as e2: last = f"re-attach failed: {e2}"
175
+ if attempt < retries - 1:
176
+ time.sleep(2); continue
177
+ raise RuntimeError(last)
149
178
  val = res.get("result", {}).get("value")
150
179
  if val is None:
151
180
  last = f"evaluate failed: {res.get('exceptionDetails')}"
@@ -175,12 +204,19 @@ FAILED = {"FAILED", "ERROR", "CANCELLED", "CANCELED"}
175
204
 
176
205
 
177
206
  def submit(session, prompt, image_urls, duration, aspect, resolution, audio,
178
- capability="reference-to-video", model="seedance-2.0-pro", last_frame_url=None):
179
- """POST the generation task. Two shapes, both seedance-2.0-pro, both URL-based (no upload):
180
- • reference-to-video: input.reference_image_urls[] + resolution
181
- • image-to-video : input.image_url (start) [+ last_frame_image_url (end)] (no resolution)
182
- Returns the task id."""
183
- if capability == "image-to-video":
207
+ capability="reference-to-video", model="seedance-2.0-pro", last_frame_url=None,
208
+ num_images=1, input_json=None):
209
+ """POST a generation task. Video + image shapes, all URL-based (no upload). Returns the task id.
210
+ • reference-to-video : reference_image_urls[] + resolution + audio
211
+ • image-to-video : image_url (start) [+ last_frame_image_url] + audio (no resolution)
212
+ • edit (Seedream) : image_url + prompt + num_images (UI hides aspect)
213
+ • text-to-image : prompt + num_images + aspect_ratio (no image)
214
+ `input_json` overrides the WHOLE input dict verbatim — for probing undocumented combos."""
215
+ medium = "image" if capability in ("edit", "text-to-image") else "video"
216
+ if input_json is not None:
217
+ inp = json.loads(input_json) # PROBE: send exactly this
218
+ meta = {"prompt": prompt, "medium": medium}
219
+ elif capability == "image-to-video":
184
220
  if not image_urls:
185
221
  raise RuntimeError("image-to-video needs a start frame (--image-url)")
186
222
  inp = {"image_url": image_urls[0], "prompt": prompt, "duration_seconds": duration, "aspect_ratio": aspect, "audio": audio}
@@ -189,6 +225,14 @@ def submit(session, prompt, image_urls, duration, aspect, resolution, audio,
189
225
  meta = {"prompt": prompt, "medium": "video", "ratio": RATIO_WORD.get(aspect, "square"), "kind": "animate", "sourceUrl": image_urls[0]}
190
226
  if last_frame_url:
191
227
  meta["lastFrameUrl"] = last_frame_url
228
+ elif capability == "edit":
229
+ if not image_urls:
230
+ raise RuntimeError("edit needs a source image (--image-url)")
231
+ inp = {"image_url": image_urls[0], "prompt": prompt, "num_images": num_images}
232
+ meta = {"prompt": prompt, "medium": "image", "kind": "edit", "sourceUrl": image_urls[0]}
233
+ elif capability == "text-to-image":
234
+ inp = {"prompt": prompt, "num_images": num_images, "aspect_ratio": aspect}
235
+ meta = {"prompt": prompt, "medium": "image", "kind": "generate"}
192
236
  else: # reference-to-video
193
237
  inp = {"reference_image_urls": image_urls, "prompt": prompt, "duration_seconds": duration, "aspect_ratio": aspect, "resolution": resolution, "audio": audio}
194
238
  meta = {"prompt": prompt, "medium": "video", "ratio": RATIO_WORD.get(aspect, "square"), "kind": "reference", "referenceImageUrls": image_urls, "referenceRequestIds": []}
@@ -224,26 +268,33 @@ def poll(session, task_id, on_status=None, timeout=1800, interval=3):
224
268
 
225
269
 
226
270
  def get_result(session, task_id):
227
- """GET the result; return (video_url, output_dict). Raises on moderation block / missing url."""
271
+ """GET the result. Returns {'videoUrl':…} for video, or {'images':[url,…]} for image. Raises on
272
+ moderation block / empty output."""
228
273
  j = session.request("GET", f"{API}/create/generation-tasks/{task_id}/result").json()
229
274
  mod = j.get("moderation_status")
230
275
  if mod and mod != "allowed":
231
276
  raise RuntimeError(f"moderation blocked: {mod}")
232
277
  out = j.get("output") or {}
233
- url = out.get("video_url")
234
- if not url:
235
- raise RuntimeError(f"no video_url in result: {json.dumps(j)[:800]}")
236
- return url, out
278
+ if out.get("video_url"):
279
+ return {"videoUrl": out["video_url"], "output": out}
280
+ imgs = [x.get("url") for x in (out.get("images") or []) if isinstance(x, dict) and x.get("url")]
281
+ if imgs:
282
+ return {"images": imgs, "output": out}
283
+ raise RuntimeError(f"no video_url/images in result: {json.dumps(j)[:800]}")
237
284
 
238
285
 
239
286
  def main():
240
287
  ap = argparse.ArgumentParser()
241
288
  ap.add_argument("--prompt", default="")
242
- ap.add_argument("--capability", default="reference-to-video", choices=["reference-to-video", "image-to-video"])
243
- ap.add_argument("--model", default="seedance-2.0-pro")
289
+ ap.add_argument("--capability", default="reference-to-video",
290
+ choices=["reference-to-video", "image-to-video", "edit", "text-to-image"])
291
+ ap.add_argument("--model", default=None, help="defaults by capability: image→seedream-5.0-pro, video→seedance-2.0-pro")
244
292
  ap.add_argument("--image-url", action="append", default=[],
245
- help="public image URL. reference-to-video: repeat for ordered refs. image-to-video: first = START frame.")
293
+ help="public image URL. reference-to-video: repeat for ordered refs. image-to-video/edit: first = start/source.")
246
294
  ap.add_argument("--last-frame-url", default=None, help="image-to-video only: END frame URL (optional)")
295
+ ap.add_argument("--num", type=int, default=1, help="num_images (image capabilities) — probe >1 here")
296
+ ap.add_argument("--input-json", default=None,
297
+ help="PROBE: raw JSON for args.input verbatim (override the built payload — test undocumented combos)")
247
298
  ap.add_argument("--duration", type=int, default=5)
248
299
  ap.add_argument("--aspect", default="1:1")
249
300
  ap.add_argument("--resolution", default="1080p")
@@ -260,6 +311,8 @@ def main():
260
311
  _JSON_OUT = sys.stdout
261
312
  sys.stdout = sys.stderr # any stray print() can't corrupt the event stream
262
313
 
314
+ # model default is capability-aware: image caps use seedream (seedANCE is the VIDEO model — a common mixup).
315
+ model = args.model or ("seedream-5.0-pro" if args.capability in ("edit", "text-to-image") else "seedance-2.0-pro")
263
316
  image_urls = args.image_url or ["https://www.gstatic.com/webp/gallery/1.jpg"] # default = permissive test image
264
317
  sess = BrowserSession(f"http://{args.host}:{args.port}", args.match)
265
318
  try:
@@ -269,11 +322,12 @@ def main():
269
322
  event(event="progress", phase="rendering", taskId=task_id)
270
323
  else:
271
324
  task_id = submit(sess, args.prompt, image_urls, args.duration, args.aspect, args.resolution, not args.no_audio,
272
- capability=args.capability, model=args.model, last_frame_url=args.last_frame_url)
325
+ capability=args.capability, model=model, last_frame_url=args.last_frame_url,
326
+ num_images=args.num, input_json=args.input_json)
273
327
  log(f"submitted task {task_id}")
274
328
  event(event="progress", phase="submitted", taskId=task_id)
275
329
  poll(sess, task_id, on_status=lambda st: (log(f" status={st}"), event(event="progress", phase=st.lower(), taskId=task_id)))
276
- video_url, out = get_result(sess, task_id)
330
+ res = get_result(sess, task_id)
277
331
  except Exception as e:
278
332
  event(event="error", reason="tryinfer", detail=str(e))
279
333
  log(f"\n❌ {e}")
@@ -281,12 +335,22 @@ def main():
281
335
  finally:
282
336
  sess.close()
283
337
 
284
- event(event="done", taskId=task_id, videoUrl=video_url,
285
- width=out.get("width"), height=out.get("height"), duration=out.get("duration_seconds"))
286
- log(f"\n✅ {out.get('width')}x{out.get('height')} {out.get('duration_seconds')}s")
287
- if not args.emit_json:
288
- print(json.dumps({"ok": True, "video_url": video_url, "width": out.get("width"),
289
- "height": out.get("height"), "duration": out.get("duration_seconds")}))
338
+ if res.get("images"):
339
+ event(event="done", taskId=task_id, images=res["images"])
340
+ dims = {x.get("url"): (x.get("width"), x.get("height")) for x in ((res.get("output") or {}).get("images") or []) if isinstance(x, dict)}
341
+ log(f"\n✅ {len(res['images'])} image(s):")
342
+ for u in res["images"]:
343
+ w, h = dims.get(u, (None, None))
344
+ log(f" {w}x{h} {u}")
345
+ if not args.emit_json:
346
+ print(json.dumps({"ok": True, "images": res["images"]}))
347
+ else:
348
+ out = res.get("output") or {}
349
+ event(event="done", taskId=task_id, videoUrl=res["videoUrl"],
350
+ width=out.get("width"), height=out.get("height"), duration=out.get("duration_seconds"))
351
+ log(f"\n✅ {out.get('width')}x{out.get('height')} {out.get('duration_seconds')}s")
352
+ if not args.emit_json:
353
+ print(json.dumps({"ok": True, "video_url": res["videoUrl"]}))
290
354
 
291
355
 
292
356
  if __name__ == "__main__":