@slatesvideo/shared 0.6.5 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/manual/content.d.ts +1 -1
- package/dist/manual/content.js +1 -1
- package/dist/operations/index.d.ts +46 -7
- package/dist/operations/index.js +169 -50
- package/dist/prompts/character-sheet.js +10 -0
- package/dist/prompts/model-capabilities.js +52 -3
- package/dist/prompts/model-facts.js +12 -4
- package/dist/prompts/prompting-tips.js +9 -3
- package/dist/prompts/reference-composer.js +14 -0
- package/dist/prompts/shot-spec.d.ts +9 -2
- package/dist/prompts/shot-spec.js +17 -1
- package/dist/skills/content.js +8 -8
- package/exports/slates-prompt-builder/generated/reference-character.md +1 -1
- package/exports/slates-prompt-builder/generated/reference-seedance.md +3 -1
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +10 -10
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +1 -1
- package/skills/slates-character-identity.md +1 -1
- package/skills/slates-model-selection.md +7 -5
- package/skills/slates-prompting-gpt-image-2-5.md +183 -0
- package/skills/slates-prompting-nano-banana-2.md +1 -1
- package/skills/slates-prompting-seedance-2-5.md +3 -1
- package/skills/slates-prompting-seedance.md +3 -1
- package/skills/slates-ugc-influencer-ad.md +5 -3
- package/skills/slates-vision-feedback-loop.md +2 -2
- package/skills/slates-prompting-gpt-image-2.md +0 -109
package/dist/operations/index.js
CHANGED
|
@@ -211,7 +211,7 @@ function creditsFromDollars(dollars) {
|
|
|
211
211
|
// Shared describe-text for the background flag on every generate_* op. ONE
|
|
212
212
|
// sentence: it is repeated verbatim on seven ops, so every word costs seven
|
|
213
213
|
// times, and `slates_get_generation_status` explains the polling itself.
|
|
214
|
-
const BACKGROUND_DESCRIBE = 'Return generationId(s)
|
|
214
|
+
const BACKGROUND_DESCRIBE = 'Return generationId(s) now instead of blocking; poll slates_get_generation_status. Recommended for video.';
|
|
215
215
|
// ── Vision QC pointers (the "quality-check with vision" rule, made structural) ──
|
|
216
216
|
//
|
|
217
217
|
// QUALITY-CHECK is a POST-condition, so it cannot be gated the way a
|
|
@@ -502,8 +502,9 @@ export const estimateGenerationCost = {
|
|
|
502
502
|
duration: z.number().int().min(1).max(360).optional().describe(`Seconds; cost scales linearly. Required with a video or PER-SECOND audio base id. Per-model windows: see slates_generate_video's duration. Audio: seed-audio ${SEED_AUDIO_MIN_SECONDS}-${SEED_AUDIO_MAX_SECONDS} (⚠️ the requested duration IS the bill), eleven-sfx ${ELEVEN_SFX_MIN_SECONDS}-${ELEVEN_SFX_MAX_SECONDS}. ⛔ NOT for ${TTS_MODEL} — pass \`characters\`.`),
|
|
503
503
|
characters: z.number().int().min(1).max(TTS_MAX_CHARACTERS).optional().describe(`${TTS_MODEL} only — the LENGTH OF THE TEXT to speak (${TTS_BUCKET_CHARS}-char buckets).`),
|
|
504
504
|
videoResolution: zEnum(VIDEO_RESOLUTIONS).optional().describe('Video only. Omitted, each model quotes at its own default. Per-model ladders: see slates_generate_video\'s videoResolution.'),
|
|
505
|
-
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k
|
|
506
|
-
quality: z.enum(['medium', 'high']).optional().describe('
|
|
505
|
+
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('Image only (default 2k; 3k: GPT Image/seedream-5-lite).'),
|
|
506
|
+
quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image tier; default high.'),
|
|
507
|
+
aspectRatio: z.string().optional().describe('Image only. 1:1/4:3/3:4 cost more than 16:9.'),
|
|
507
508
|
sound: z.boolean().optional().describe('Veo only — audio flag changes the cost key.'),
|
|
508
509
|
seedanceFace: z.boolean().optional().describe('Seedance AI-face route (pricier key).'),
|
|
509
510
|
seedanceRealFace: z.boolean().optional().describe('Seedance consented real-face route (premium key).'),
|
|
@@ -514,15 +515,21 @@ export const estimateGenerationCost = {
|
|
|
514
515
|
const byKey = new Map(registry.models.map((m) => [m.model, creditCost(m)]));
|
|
515
516
|
// 1) exact registry cost key
|
|
516
517
|
let key = byKey.has(input.model) ? input.model : null;
|
|
517
|
-
// 2) image base id + resolution (+ quality for
|
|
518
|
+
// 2) image base id + resolution (+ quality for GPT Image 2.5)
|
|
518
519
|
if (!key) {
|
|
519
520
|
// IMAGE_MODELS, never a second hand-typed copy: this list is declared
|
|
520
521
|
// below (a runtime read, so no temporal-dead-zone hazard) and is the same
|
|
521
522
|
// enum `slates_generate_image` accepts. Two copies is how the estimate op
|
|
522
523
|
// would quietly stop pricing the seventh image model.
|
|
523
524
|
const img = IMAGE_MODELS.find((m) => m === input.model);
|
|
525
|
+
// ⚙ NO INLINE DEFAULT. `imageCostKey`'s own parameter default is the ONE
|
|
526
|
+
// home for the fallback tier, and `pricing-consistency-check.mjs` pins it
|
|
527
|
+
// to the desktop's. This line read `?? 'medium'` while every generate path
|
|
528
|
+
// billed `high`, so the op the doctrine tells agents to call before every
|
|
529
|
+
// generation quoted 1 cr for a 2 cr job. Four separate copies of one
|
|
530
|
+
// default is what made that possible; there are now none.
|
|
524
531
|
if (img)
|
|
525
|
-
key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality
|
|
532
|
+
key = imageCostKey(img, input.resolution ?? (img === 'nano-banana-2-lite' ? '1k' : '2k'), input.quality, input.aspectRatio);
|
|
526
533
|
}
|
|
527
534
|
// 2a) audio base id → seconds. Both surfaces bill per second, so a
|
|
528
535
|
// duration is always required. Runs BEFORE the video resolver: it is
|
|
@@ -642,7 +649,7 @@ export const estimateGenerationCost = {
|
|
|
642
649
|
if (key == null || perCredits == null) {
|
|
643
650
|
// Every id in the error comes from the SSOT arrays. The image half was
|
|
644
651
|
// hand-typed and named three of six, so an agent that mis-spelled
|
|
645
|
-
// `gpt-image-2` was told the model did not exist.
|
|
652
|
+
// `gpt-image-2-5-flare` was told the model did not exist.
|
|
646
653
|
throw new Error(`Unknown model: ${input.model}. Pass a base id (${VIDEO_MODELS.join(' | ')} | ${AUDIO_MODELS.join(' | ')} | ${IMAGE_MODELS.join(' | ')}) plus duration/resolution params, or use slates_list_available_models with a filter.`);
|
|
647
654
|
}
|
|
648
655
|
const qty = input.quantity ?? 1;
|
|
@@ -1123,11 +1130,23 @@ export const generateCharacterIdentity = {
|
|
|
1123
1130
|
baseAssetId: z.string().uuid().describe('The base portrait asset the identity is generated from.'),
|
|
1124
1131
|
userNotes: z.string().optional().describe('Extra instruction, e.g. "use the woman on the left".'),
|
|
1125
1132
|
model: z
|
|
1126
|
-
.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2'])
|
|
1133
|
+
.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2-5-flare', 'gpt-image-2-5-sunburst'])
|
|
1127
1134
|
.optional()
|
|
1128
1135
|
.describe('Image model for the sheet. Omit for the default (nano-banana-2). Exists so the layout-vs-face tradeoff can be tested with comparison gens — do not switch without a receipt.'),
|
|
1129
1136
|
}),
|
|
1130
1137
|
async run(input, ctx) {
|
|
1138
|
+
// 🚨 THE ROSTER GATE, WHICH THIS OP NEVER HAD. Its `model` enum offers
|
|
1139
|
+
// seats an older desktop does not know, and the handler resolves an unknown
|
|
1140
|
+
// id by falling back to the Google/Banana path — so the op would quote one
|
|
1141
|
+
// model's price and generate another, silently. Same failure `generateImage`
|
|
1142
|
+
// and `editImage` gate against; this op was simply missed when the gate was
|
|
1143
|
+
// introduced, and the 2.5 swap is what makes it reachable in practice.
|
|
1144
|
+
if (isGptImageModel(input.model)) {
|
|
1145
|
+
await ctx.desktop().requireCapability('image-models-v3', `${input.model} character identity`);
|
|
1146
|
+
}
|
|
1147
|
+
else if (input.model === 'nano-banana-pro' || input.model === 'nano-banana-2-lite') {
|
|
1148
|
+
await ctx.desktop().requireCapability('image-models-v2', `${input.model} character identity`);
|
|
1149
|
+
}
|
|
1131
1150
|
return ok(await ctx.desktop().post('/agent/characters/generate-identity', {
|
|
1132
1151
|
characterId: input.characterId,
|
|
1133
1152
|
projectId: input.projectId,
|
|
@@ -1309,21 +1328,39 @@ async function previewAssets(ctx, refs) {
|
|
|
1309
1328
|
return out;
|
|
1310
1329
|
}
|
|
1311
1330
|
/** The exact `model` ids `slates_generate_image` accepts. */
|
|
1331
|
+
/**
|
|
1332
|
+
* The image models whose ladder includes the 3k (1440p) class — a MIRROR of
|
|
1333
|
+
* `imageResolutions` in slate's MODEL_REGISTRY, which this package cannot read.
|
|
1334
|
+
* Exported so `pricing-consistency-check.mjs` can prove the mirror still
|
|
1335
|
+
* matches; without that proof a model that gains 3k in the registry just goes
|
|
1336
|
+
* quietly unreachable through the op.
|
|
1337
|
+
*/
|
|
1338
|
+
export const THREE_K_IMAGE_MODELS = [
|
|
1339
|
+
'gpt-image-2-5-flare',
|
|
1340
|
+
'gpt-image-2-5-sunburst',
|
|
1341
|
+
'seedream-5-lite',
|
|
1342
|
+
];
|
|
1312
1343
|
export const IMAGE_MODELS = [
|
|
1313
1344
|
'nano-banana-2',
|
|
1314
1345
|
'nano-banana-2-lite',
|
|
1315
1346
|
'nano-banana-pro',
|
|
1316
|
-
'gpt-image-2',
|
|
1347
|
+
'gpt-image-2-5-flare',
|
|
1348
|
+
'gpt-image-2-5-sunburst',
|
|
1317
1349
|
'flux-2-max',
|
|
1318
1350
|
'seedream-5-lite',
|
|
1319
1351
|
];
|
|
1352
|
+
/** True for either GPT Image 2.5 seat. Flare and Sunburst differ in latency
|
|
1353
|
+
* and routing advice, never in pricing shape or param surface. */
|
|
1354
|
+
export function isGptImageModel(model) {
|
|
1355
|
+
return model === 'gpt-image-2-5-flare' || model === 'gpt-image-2-5-sunburst';
|
|
1356
|
+
}
|
|
1320
1357
|
/**
|
|
1321
1358
|
* The aspect ratios an image generation can carry — the UNION over what these
|
|
1322
1359
|
* models declare in `MODEL_CAPABILITIES`, generated so the enum cannot hold a
|
|
1323
1360
|
* value no model accepts. It used to be a hand-typed eleven-value list whose
|
|
1324
1361
|
* `9:21` exists in ZERO models; that phantom is gone by construction.
|
|
1325
1362
|
*
|
|
1326
|
-
* ⚠️ STILL A UNION, NOT A PER-MODEL CHECK.
|
|
1363
|
+
* ⚠️ STILL A UNION, NOT A PER-MODEL CHECK. GPT Image 2.5 takes five of these
|
|
1327
1364
|
* ten and the other five would be accepted here. The image param surface has
|
|
1328
1365
|
* not been audited (aspect ratios, resolution classes, per-model reference
|
|
1329
1366
|
* caps) — that audit is the named follow-up in
|
|
@@ -1335,9 +1372,35 @@ const IMAGE_ASPECT_RATIOS = aspectRatioUnion(IMAGE_MODELS);
|
|
|
1335
1372
|
// Registry cost-key for an image model+resolution. Mirrors imageCreditKey()
|
|
1336
1373
|
// in slate/src/shared/pricing.ts — MUST byte-match it (the same hard rule as
|
|
1337
1374
|
// videoCostKey): NB2/NB Pro price per resolution, FLUX.2 Max prices per
|
|
1338
|
-
// resolution (1k is the bare key), NB2 Lite/Seedream are flat, GPT Image 2 is
|
|
1339
|
-
// quality × resolution-class
|
|
1340
|
-
|
|
1375
|
+
// resolution (1k is the bare key), NB2 Lite/Seedream are flat, GPT Image 2.5 is
|
|
1376
|
+
// quality × resolution-class over ALL FIVE exposed tiers.
|
|
1377
|
+
//
|
|
1378
|
+
// 🚨 THE KEY SUFFIX IS `med` WHILE THE WIRE VALUE IS `medium` — the one
|
|
1379
|
+
// tier name that differs between fal's enum and our cost keys, inherited from
|
|
1380
|
+
// GPT Image 2 (`gpt-image-2-med-4k`). `gptKeyTier` is that whole translation,
|
|
1381
|
+
// and it must stay byte-identical to `gptKeyTier` in
|
|
1382
|
+
// slate/src/shared/pricing.ts — pricing-consistency-check.mjs enforces it.
|
|
1383
|
+
function gptKeyTier(quality) {
|
|
1384
|
+
return quality === 'medium' ? 'med' : quality;
|
|
1385
|
+
}
|
|
1386
|
+
// 🚨 THE ASPECT RATIO IS PART OF THE PRICE ON GPT IMAGE. fal bills image
|
|
1387
|
+
// OUTPUT TOKENS and the count tracks the frame's SHAPE — metered 2026-09-09,
|
|
1388
|
+
// 4:3/3:4 cost ~4/3 of the class rate and 1:1 ~16/9 of it. Quoting one price
|
|
1389
|
+
// for every aspect sold 1:1 below cost at `high` and above. Byte-identical to
|
|
1390
|
+
// `gptKeyAspect` in slate/src/shared/pricing.ts; pricing-consistency-check
|
|
1391
|
+
// proves it key by key. 16:9 and 9:16 keep the bare key they always had.
|
|
1392
|
+
function gptKeyAspect(aspectRatio) {
|
|
1393
|
+
if (aspectRatio === '1:1')
|
|
1394
|
+
return '-sq';
|
|
1395
|
+
if (aspectRatio === '4:3' || aspectRatio === '3:4')
|
|
1396
|
+
return '-43';
|
|
1397
|
+
return '';
|
|
1398
|
+
}
|
|
1399
|
+
// EXPORTED for the same reason `videoCostKey` is: `pricing-consistency-check.mjs`
|
|
1400
|
+
// imports it and asserts, key by key, that it equals the desktop's
|
|
1401
|
+
// `imageCreditKey`. Before that check existed the byte-match was a comment and
|
|
1402
|
+
// a hope — the two are in different repos and nothing compared them.
|
|
1403
|
+
export function imageCostKey(model, resolution, quality = 'high', aspectRatio) {
|
|
1341
1404
|
if (model === 'flux-2-max')
|
|
1342
1405
|
return resolution === '1k' ? 'flux-2-max' : `flux-2-max-${resolution}`;
|
|
1343
1406
|
if (model === 'seedream-5-lite')
|
|
@@ -1346,8 +1409,8 @@ function imageCostKey(model, resolution, quality = 'medium') {
|
|
|
1346
1409
|
return 'nano-banana-2-lite';
|
|
1347
1410
|
if (model === 'nano-banana-pro')
|
|
1348
1411
|
return `nano-banana-pro-${resolution}`;
|
|
1349
|
-
if (model
|
|
1350
|
-
return
|
|
1412
|
+
if (isGptImageModel(model))
|
|
1413
|
+
return `${model}-${gptKeyTier(quality)}-${resolution}${gptKeyAspect(aspectRatio)}`;
|
|
1351
1414
|
return `nano-banana-2-${resolution}`;
|
|
1352
1415
|
}
|
|
1353
1416
|
export const generateImage = {
|
|
@@ -1359,9 +1422,9 @@ export const generateImage = {
|
|
|
1359
1422
|
// (it still described nano-banana-2-lite by a capability the param owns).
|
|
1360
1423
|
`${describeRouting('image')}\n` +
|
|
1361
1424
|
'Full table: the slates-model-selection skill. ' +
|
|
1362
|
-
'Pass projectId to save into a Slates project (
|
|
1425
|
+
'Pass projectId to save into a Slates project (asset appears live in the desktop UI). All models except nano-banana-2 REQUIRE projectId (no headless path). REQUIRED before calling: read the slates-cost-discipline skill (and the model\'s slates-prompting-* skill). You MUST pass aspectRatio and resolution explicitly (the server returns requires_clarification when missing — defaults waste credits). ' +
|
|
1363
1426
|
CONFIRM_GATE_SENTENCE +
|
|
1364
|
-
' MCP/CLI generation always charges credits. No
|
|
1427
|
+
' MCP/CLI generation always charges credits. No skills installed? Call slates_get_prompting_guide with the model\'s topic and \'slates-cost-discipline\' first. ' +
|
|
1365
1428
|
// GENERATED from the skill file's own never-use list -- the one piece of
|
|
1366
1429
|
// prompting doctrine that is ALWAYS in context, because the agent has
|
|
1367
1430
|
// demonstrably skipped the call that would have taught it.
|
|
@@ -1370,12 +1433,13 @@ export const generateImage = {
|
|
|
1370
1433
|
prompt: z.string().min(1).max(4000),
|
|
1371
1434
|
model: zEnum(IMAGE_MODELS).optional().describe('Image model. Default nano-banana-2. Routing doctrine: slates-model-selection skill. All except nano-banana-2 require projectId.'),
|
|
1372
1435
|
projectId: z.string().uuid().optional().describe('Save into this Slates project. Renderer refreshes live. Required for every model except nano-banana-2.'),
|
|
1373
|
-
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('
|
|
1374
|
-
quality: z.enum(['medium', 'high']).optional().describe('
|
|
1375
|
-
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1436
|
+
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('1k drafts, 2k hero, 4k final. nano-banana-2-lite: 1k only. GPT Image classes 1024²/1080p/1440p/2160p. Never default this.'),
|
|
1437
|
+
quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image only. UNEVEN ladder: max=4× high, xhigh~1.8×. medium drafts; default high.'),
|
|
1438
|
+
backgroundMode: z.enum(['auto', 'transparent', 'opaque']).optional().describe('GPT Image only. transparent = alpha channel. Free.'),
|
|
1439
|
+
aspectRatio: zEnum(IMAGE_ASPECT_RATIOS).optional().describe(`Pick from the use case: cinematic 16:9 · TikTok/Reels 9:16 · IG square 1:1 · ultra-wide 21:9. 1:1 costs most on GPT Image. Per model: ${describeAspectRatios(IMAGE_MODELS)}`),
|
|
1440
|
+
count: z.number().int().min(1).max(10).optional().describe('Up to 10 with projectId; headless caps at 4.'),
|
|
1441
|
+
referenceImageUrls: z.array(z.string().url()).max(14).optional().describe('Headless (no projectId) nano-banana-2 only. With a projectId, upload via slates_upload_reference_image. Label every image role in the prompt.'),
|
|
1442
|
+
referenceAssetIds: z.array(z.string()).max(16).optional().describe("Project assets as references — UUIDs or badge codes (\"IMG-A8\"), resolved at call time. Requires projectId. Caps: GPT Image 16, nano-banana-2 14, FLUX/Seedream lower. Label every reference role in the prompt."),
|
|
1379
1443
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
1380
1444
|
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate.'),
|
|
1381
1445
|
}),
|
|
@@ -1408,13 +1472,24 @@ export const generateImage = {
|
|
|
1408
1472
|
}
|
|
1409
1473
|
const resolution = input.resolution;
|
|
1410
1474
|
const imageModel = input.model ?? 'nano-banana-2';
|
|
1411
|
-
//
|
|
1412
|
-
//
|
|
1413
|
-
|
|
1475
|
+
// 🚨 SEEDREAM HAS A 3k CLASS TOO, AND THIS GUARD USED TO DENY IT.
|
|
1476
|
+
// It read `!== 'gpt-image-2'` and rejected every other model at 3k — but
|
|
1477
|
+
// `seedream-5-lite` declares ['2k','3k','4k'] in the desktop registry and
|
|
1478
|
+
// has a real `seedream-5-lite` cost key, so the op was refusing a request
|
|
1479
|
+
// the desktop would have served. Pre-existing; found by the 2026-09-09
|
|
1480
|
+
// audit, not introduced by the 2.5 swap.
|
|
1481
|
+
//
|
|
1482
|
+
// The ladder itself is owned by MODEL_REGISTRY in slate/src/shared/pricing.ts
|
|
1483
|
+
// and is not readable from here, so THREE_K_IMAGE_MODELS is a MIRROR — declared
|
|
1484
|
+
// and exported below so `pricing-consistency-check.mjs` compares it against
|
|
1485
|
+
// the desktop registry's own `imageResolutions`. It used to be an inline
|
|
1486
|
+
// literal with a comment admitting nothing checked it, which is how it came
|
|
1487
|
+
// to deny `seedream-5-lite` a class the desktop had always served.
|
|
1488
|
+
if (resolution === '3k' && !THREE_K_IMAGE_MODELS.includes(imageModel)) {
|
|
1414
1489
|
return ok({
|
|
1415
1490
|
requires_clarification: true,
|
|
1416
1491
|
missing: ['resolution'],
|
|
1417
|
-
message: `3k (1440p)
|
|
1492
|
+
message: `3k (1440p) exists on ${THREE_K_IMAGE_MODELS.join(', ')} — pick 1k/2k/4k for ${imageModel}.`,
|
|
1418
1493
|
});
|
|
1419
1494
|
}
|
|
1420
1495
|
// Only nano-banana-2 has a headless path — everything else routes through
|
|
@@ -1445,6 +1520,18 @@ export const generateImage = {
|
|
|
1445
1520
|
message: 'background=true routes through the desktop generation pipeline (so slates_get_generation_status can poll it) — pass a projectId, or drop background for a blocking headless run.',
|
|
1446
1521
|
});
|
|
1447
1522
|
}
|
|
1523
|
+
// 🚨 THE HEADLESS PATH ASKS FAL FOR A BATCH, so a provider ceiling binds
|
|
1524
|
+
// here and nowhere else. It is nano-banana-2 only, and Nano Banana's
|
|
1525
|
+
// `num_images` maximum is 4 (fal schema, 2026-09-09). Refused rather than
|
|
1526
|
+
// clamped: a silent clamp would make four images against a request for ten
|
|
1527
|
+
// and read to the caller as a partial failure it should retry.
|
|
1528
|
+
if (!input.projectId && (input.count ?? 1) > 4) {
|
|
1529
|
+
return ok({
|
|
1530
|
+
requires_clarification: true,
|
|
1531
|
+
missing: ['projectId'],
|
|
1532
|
+
message: 'count above 4 needs a projectId. The headless path asks fal for one batch and nano-banana-2 caps a batch at 4; with a projectId the desktop fires them as separate generations and the limit is 10.',
|
|
1533
|
+
});
|
|
1534
|
+
}
|
|
1448
1535
|
let refEcho = '';
|
|
1449
1536
|
if (referenceAssetIds.length > 0) {
|
|
1450
1537
|
await ctx.desktop().requireCapability('image-references', 'reference images on image generation');
|
|
@@ -1461,11 +1548,18 @@ export const generateImage = {
|
|
|
1461
1548
|
// New-roster models need a desktop that knows them — an older desktop's
|
|
1462
1549
|
// allowlist would silently fall back to nano-banana-2 while we quote the
|
|
1463
1550
|
// new model's price.
|
|
1464
|
-
if (input.projectId &&
|
|
1465
|
-
|
|
1551
|
+
if (input.projectId && isGptImageModel(imageModel)) {
|
|
1552
|
+
// 🚨 v3, NOT v2. GPT Image 2.5 landed 2026-09-09 with new ids and a
|
|
1553
|
+
// five-rung ladder; a desktop that only knows v2 has neither, so it would
|
|
1554
|
+
// fall back to nano-banana-2 while this op quotes a 2.5 price — precisely
|
|
1555
|
+
// the failure the gate was built to stop. Reusing v2 reintroduces it.
|
|
1556
|
+
await ctx.desktop().requireCapability('image-models-v3', `${imageModel} generation`);
|
|
1557
|
+
}
|
|
1558
|
+
else if (input.projectId &&
|
|
1559
|
+
(imageModel === 'nano-banana-pro' || imageModel === 'nano-banana-2-lite')) {
|
|
1466
1560
|
await ctx.desktop().requireCapability('image-models-v2', `${imageModel} generation`);
|
|
1467
1561
|
}
|
|
1468
|
-
const costKey = imageCostKey(imageModel, resolution, input.quality ?? '
|
|
1562
|
+
const costKey = imageCostKey(imageModel, resolution, input.quality, input.aspectRatio ?? '1:1');
|
|
1469
1563
|
const cloud = ctx.cloud();
|
|
1470
1564
|
const registry = await cloud.get('/api/agent/models');
|
|
1471
1565
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
@@ -1534,7 +1628,7 @@ export const generateImage = {
|
|
|
1534
1628
|
resolution,
|
|
1535
1629
|
aspectRatio: input.aspectRatio ?? '1:1',
|
|
1536
1630
|
count: input.count ?? 1,
|
|
1537
|
-
...(imageModel
|
|
1631
|
+
...(isGptImageModel(imageModel) ? { gptQuality: input.quality, gptBackground: input.backgroundMode } : {}),
|
|
1538
1632
|
...(referenceAssetIds.length > 0 ? { referenceAssetIds } : {}),
|
|
1539
1633
|
background: input.background,
|
|
1540
1634
|
});
|
|
@@ -1627,6 +1721,13 @@ export const generateImage = {
|
|
|
1627
1721
|
params: {
|
|
1628
1722
|
prompt: input.prompt,
|
|
1629
1723
|
aspect_ratio: input.aspectRatio ?? '1:1',
|
|
1724
|
+
// 🚨 THE HEADLESS PATH IS THE ONE PLACE WE ASK FAL FOR A BATCH, so it
|
|
1725
|
+
// is the one place a provider's own `num_images` ceiling binds — and
|
|
1726
|
+
// Nano Banana's is 4 (fal schema, read 2026-09-09), against the op's
|
|
1727
|
+
// limit of 10. Everything else fans out through the desktop as N
|
|
1728
|
+
// separate single-image generations, where no batch ceiling exists.
|
|
1729
|
+
// Guarded above rather than clamped here: silently making 4 when 10
|
|
1730
|
+
// were asked for would bill 4 and look like a partial failure.
|
|
1630
1731
|
num_images: input.count ?? 1,
|
|
1631
1732
|
...(hasReferenceImages
|
|
1632
1733
|
? { image_urls: input.referenceImageUrls }
|
|
@@ -1697,15 +1798,16 @@ async function pollProxyJob(cloud, jobId, options = {}) {
|
|
|
1697
1798
|
export const editImage = {
|
|
1698
1799
|
id: 'slates_edit_image',
|
|
1699
1800
|
billable: true,
|
|
1700
|
-
description: 'Surgically edit an
|
|
1801
|
+
description: 'Surgically edit an image asset with a text instruction (e.g. \'make the jacket red\') instead of regenerating from scratch — use when ~90% of the image is already right. The result is a NEW asset (prompt prefixed \'[Edit]\'); the source is untouched. Default model nano-banana-2 (only model that also accepts referenceAssetIds); flux-2-max / seedream-5-lite use their own edit endpoints and ignore references. Before first use call slates_get_prompting_guide with topic \'slates-edit-and-iterate\'.',
|
|
1701
1802
|
input: z.object({
|
|
1702
1803
|
projectId: z.string().uuid(),
|
|
1703
|
-
sourceAssetId: z.string().uuid().describe('Image asset to edit. Must
|
|
1704
|
-
prompt: z.string().min(1).max(4000).describe('The
|
|
1705
|
-
editModel: z.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2', 'flux-2-max', 'seedream-5-lite']).optional(),
|
|
1706
|
-
referenceAssetIds: z.array(z.string().uuid()).max(13).optional().describe('Nano-Banana
|
|
1707
|
-
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('3k
|
|
1708
|
-
quality: z.enum(['medium', 'high']).optional().describe('
|
|
1804
|
+
sourceAssetId: z.string().uuid().describe('Image asset to edit. Must exist in the project.'),
|
|
1805
|
+
prompt: z.string().min(1).max(4000).describe('The change, not the whole image.'),
|
|
1806
|
+
editModel: z.enum(['nano-banana-2', 'nano-banana-2-lite', 'nano-banana-pro', 'gpt-image-2-5-flare', 'gpt-image-2-5-sunburst', 'flux-2-max', 'seedream-5-lite']).optional(),
|
|
1807
|
+
referenceAssetIds: z.array(z.string().uuid()).max(13).optional().describe('Nano-Banana only (NB Pro 13, NB2 Lite 3).'),
|
|
1808
|
+
resolution: z.enum(['1k', '2k', '3k', '4k']).optional().describe('3k = GPT Image/seedream-5-lite; nano-banana-2-lite is 1k only.'),
|
|
1809
|
+
quality: z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional().describe('GPT Image tier; default high.'),
|
|
1810
|
+
backgroundMode: z.enum(['auto', 'transparent', 'opaque']).optional().describe('GPT Image only. transparent = alpha channel. Free.'),
|
|
1709
1811
|
aspectRatio: z.string().optional(),
|
|
1710
1812
|
confirm: z.boolean().optional().describe('Set true to bypass the confirm gate.'),
|
|
1711
1813
|
background: z.boolean().optional().describe(BACKGROUND_DESCRIBE),
|
|
@@ -1718,14 +1820,22 @@ export const editImage = {
|
|
|
1718
1820
|
}
|
|
1719
1821
|
const editModel = input.editModel ?? 'nano-banana-2';
|
|
1720
1822
|
const resolution = input.resolution ?? (editModel === 'nano-banana-2-lite' ? '1k' : '2k');
|
|
1721
|
-
if ((editModel
|
|
1823
|
+
if (isGptImageModel(editModel)) {
|
|
1824
|
+
// 🚨 v3, LIKE generateImage — this site was missed once already.
|
|
1825
|
+
// A pre-2.5 desktop advertises v2, so gating the 2.5 seats on v2 lets it
|
|
1826
|
+
// through; `isFalImageModel` is then false for these ids on that build and
|
|
1827
|
+
// handleEditImage falls through to the nano-banana path — NB2 output billed
|
|
1828
|
+
// at a quoted 2.5 price. Exactly the bug the gate exists to stop.
|
|
1829
|
+
await desktop.requireCapability('image-models-v3', `${editModel} editing`);
|
|
1830
|
+
}
|
|
1831
|
+
else if (editModel === 'nano-banana-pro' || editModel === 'nano-banana-2-lite') {
|
|
1722
1832
|
await desktop.requireCapability('image-models-v2', `${editModel} editing`);
|
|
1723
1833
|
}
|
|
1724
|
-
// Nano-Banana family + GPT Image 2 edits charge the same key as gen;
|
|
1834
|
+
// Nano-Banana family + GPT Image 2.5 edits charge the same key as gen;
|
|
1725
1835
|
// FLUX / Seedream route to dedicated edit endpoints priced under '-edit' keys.
|
|
1726
1836
|
const costKey = editModel === 'flux-2-max' || editModel === 'seedream-5-lite'
|
|
1727
1837
|
? `${imageCostKey(editModel, resolution)}-edit`
|
|
1728
|
-
: imageCostKey(editModel, resolution, input.quality
|
|
1838
|
+
: imageCostKey(editModel, resolution, input.quality, input.aspectRatio);
|
|
1729
1839
|
const cloud = ctx.cloud();
|
|
1730
1840
|
const registry = await cloud.get('/api/agent/models');
|
|
1731
1841
|
const entry = registry.models.find((m) => m.model === costKey);
|
|
@@ -1752,7 +1862,7 @@ export const editImage = {
|
|
|
1752
1862
|
editModel,
|
|
1753
1863
|
referenceAssetIds: input.referenceAssetIds,
|
|
1754
1864
|
resolution,
|
|
1755
|
-
...(editModel
|
|
1865
|
+
...(isGptImageModel(editModel) ? { gptQuality: input.quality, gptBackground: input.backgroundMode } : {}),
|
|
1756
1866
|
aspectRatio: input.aspectRatio,
|
|
1757
1867
|
background: input.background,
|
|
1758
1868
|
});
|
|
@@ -4105,13 +4215,14 @@ function shotParamsShape(described) {
|
|
|
4105
4215
|
return {
|
|
4106
4216
|
aspectRatio: d(zEnum(SHOT_ASPECT_RATIOS).optional(), 'Validated against the model; see slates_generate_video.'),
|
|
4107
4217
|
duration: d(z.number().int().min(1).max(360).optional(), 'Seconds for video or duration-based audio; TTS uses text length.'),
|
|
4108
|
-
videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the
|
|
4218
|
+
videoResolution: d(zEnum(VIDEO_RESOLUTIONS).optional(), 'Validated against the model when the Shot is saved.'),
|
|
4109
4219
|
imageResolution: d(z.enum(['1k', '2k', '3k', '4k']).optional(), 'Image models only.'),
|
|
4110
|
-
gptQuality: d(z.enum(['medium', 'high']).optional(), '
|
|
4111
|
-
|
|
4220
|
+
gptQuality: d(z.enum(['low', 'medium', 'high', 'xhigh', 'max']).optional(), 'GPT Image 2.5 only.'),
|
|
4221
|
+
gptBackground: d(z.enum(['auto', 'transparent', 'opaque']).optional(), 'GPT Image only.'),
|
|
4222
|
+
imageQuantity: d(z.number().int().min(1).max(10).optional(), 'Image models only.'),
|
|
4112
4223
|
negativePrompt: z.string().optional(),
|
|
4113
4224
|
sound: d(z.boolean().optional(), 'Video models that co-generate audio.'),
|
|
4114
|
-
seedanceFace: d(z.boolean().optional(), "Seedance only — a reference shows an AI character's FACE; reroutes to a face-capable provider
|
|
4225
|
+
seedanceFace: d(z.boolean().optional(), "Seedance only — a reference shows an AI character's FACE; reroutes to a face-capable provider, ~45% more."),
|
|
4115
4226
|
audioDurationSeconds: d(z.number().int().min(1).max(120).optional(), 'Audio lane. On seed-audio the requested duration IS the bill.'),
|
|
4116
4227
|
// The TTS voice — the same three fields slates_generate_audio takes, so a
|
|
4117
4228
|
// Shot is the audio call, serialized. Exactly one of them, enforced at fire.
|
|
@@ -4364,7 +4475,15 @@ function shotCostKey(detail) {
|
|
|
4364
4475
|
}
|
|
4365
4476
|
if (IMAGE_MODELS.includes(model)) {
|
|
4366
4477
|
return imageCostKey(model, (fires?.imageResolution ?? p.imageResolution) ??
|
|
4367
|
-
(model === 'nano-banana-2-lite' ? '1k' : '2k'),
|
|
4478
|
+
(model === 'nano-banana-2-lite' ? '1k' : '2k'),
|
|
4479
|
+
// No inline default — see the note at the estimate op. Firing a Shot with
|
|
4480
|
+
// no stored tier lands on `high` via the desktop's `normalizeGptQuality`,
|
|
4481
|
+
// and `imageCostKey`'s parameter default is pinned to match it.
|
|
4482
|
+
p.gptQuality,
|
|
4483
|
+
// The aspect the Shot will fire at — on GPT Image it moves the key, so
|
|
4484
|
+
// quoting without it under-prices every square Shot. `fires` carries
|
|
4485
|
+
// only the clamped resolution/duration axes, never the aspect.
|
|
4486
|
+
p.aspectRatio);
|
|
4368
4487
|
}
|
|
4369
4488
|
return null;
|
|
4370
4489
|
}
|
|
@@ -4663,8 +4782,8 @@ export const listShots = {
|
|
|
4663
4782
|
' depend on the reference set — Seedance reference-clip seconds and MiniMax reference images' +
|
|
4664
4783
|
' past the free five — are missing from it. slates_get_shot prices one exactly, and' +
|
|
4665
4784
|
' slates_generate_from_shots quotes the set exactly before it fires anything.' +
|
|
4666
|
-
(describeVarietyReport(r.variety) ? `
|
|
4667
|
-
|
|
4785
|
+
(describeVarietyReport(r.variety) ? `
|
|
4786
|
+
|
|
4668
4787
|
${describeVarietyReport(r.variety)}` : ''));
|
|
4669
4788
|
},
|
|
4670
4789
|
};
|
|
@@ -4861,7 +4980,7 @@ function resolveGuideTopic(topic) {
|
|
|
4861
4980
|
if (t.startsWith('nano-banana'))
|
|
4862
4981
|
return 'slates-prompting-nano-banana-2';
|
|
4863
4982
|
if (t.startsWith('gpt-image') || t.startsWith('gpt image'))
|
|
4864
|
-
return 'slates-prompting-gpt-image-2';
|
|
4983
|
+
return 'slates-prompting-gpt-image-2-5';
|
|
4865
4984
|
if (t.startsWith('flux'))
|
|
4866
4985
|
return 'slates-prompting-flux-2-max';
|
|
4867
4986
|
if (t.startsWith('seedream'))
|
|
@@ -4953,7 +5072,7 @@ function resolveGuideTopic(topic) {
|
|
|
4953
5072
|
* The guide index, GENERATED from SKILLS.
|
|
4954
5073
|
*
|
|
4955
5074
|
* The list here was hand-typed and had drifted to 25 of 32 names — the prompting
|
|
4956
|
-
* guides for GPT Image
|
|
5075
|
+
* guides for GPT Image, MiniMax H3, LTX-2.5, Seedance 2.5 and Omni Flash were
|
|
4957
5076
|
* all missing, so an agent reading this description could not learn they exist.
|
|
4958
5077
|
* A hand-typed index of a generated corpus is a stale index; it is only a matter
|
|
4959
5078
|
* of when.
|
|
@@ -78,6 +78,16 @@
|
|
|
78
78
|
// without the absence clause. If that happens, the fix is a
|
|
79
79
|
// model-conditional phrasing, not restoring the 422.
|
|
80
80
|
//
|
|
81
|
+
// 2026-09-09 — THE MODEL IT WAS MEASURED ON IS RETIRED; THE RULE IS NOT.
|
|
82
|
+
// GPT Image 2.5 replaced gpt-image-2 in the picker. The 422 above was
|
|
83
|
+
// measured on gpt-image-2 and that wording is left exactly as recorded,
|
|
84
|
+
// because a receipt names what was actually tested. What carries over is
|
|
85
|
+
// the MECHANISM, not the measurement: the classifier belongs to OpenAI,
|
|
86
|
+
// not to a model version, so phrase exclusions as framing on 2.5 too.
|
|
87
|
+
// ⚠️ INHERITED, NOT RE-MEASURED — nobody has re-run the 422 on Flare or
|
|
88
|
+
// Sunburst. If one of them accepts the absence clause, that is a new
|
|
89
|
+
// receipt to write down, not a reason to delete this one.
|
|
90
|
+
//
|
|
81
91
|
// 2026-07-30 (b) — THE GENRE ANCHOR HAS TO BE SCOPED TO THE FACE, and
|
|
82
92
|
// this one cost real generations. "an invisible-mannequin presentation
|
|
83
93
|
// WHERE THE CLOTHING HOLDS ITS OWN SHAPE" is the e-commerce genre stated
|
|
@@ -133,10 +133,48 @@ export const MODEL_CAPABILITIES = {
|
|
|
133
133
|
aspectRatios: FULL_ASPECT_RATIOS,
|
|
134
134
|
maxRefImages: 14,
|
|
135
135
|
},
|
|
136
|
-
|
|
137
|
-
|
|
136
|
+
// `gpt-image-2` WAS HERE AND IS DELIBERATELY GONE (retired 2026-09-09).
|
|
137
|
+
//
|
|
138
|
+
// An earlier pass kept the row so that a Shot saved before the swap could
|
|
139
|
+
// still resolve its caps by stored id. That was the wrong fix and the
|
|
140
|
+
// desktop refuses it: `pricing.ts` throws at MODULE LOAD for any capability
|
|
141
|
+
// row with no MODEL_REGISTRY entry, because an orphan row makes THIS op
|
|
142
|
+
// advertise, validate and quote a model the desktop can no longer render —
|
|
143
|
+
// the agent passes every gate and then hits 'Unsupported model' at the
|
|
144
|
+
// handler.
|
|
145
|
+
//
|
|
146
|
+
// Old Shots are handled where they are READ instead: `migrateGptImageModel`
|
|
147
|
+
// in slate/src/shared/pricing.ts rewrites the stored id to Flare inside
|
|
148
|
+
// `transform` (slate/src/main/storage/shots.ts), the one place a DB row
|
|
149
|
+
// becomes a Shot. That is strictly better than keeping the row — the Shot
|
|
150
|
+
// comes back FIREABLE on a live model, rather than merely openable on a dead
|
|
151
|
+
// one. Do not re-add this row to make a stale id resolve; migrate it.
|
|
152
|
+
// GPT Image 2.5 — 16 references, which IS fal's documented ceiling rather
|
|
153
|
+
// than a number of ours: `image_urls` carries `maxItems: 16` on both 2.5
|
|
154
|
+
// endpoints AND on both gpt-image-2 endpoints (schema, read 2026-09-09,
|
|
155
|
+
// ripped verbatim to second-brain/business/projects/slates/research/
|
|
156
|
+
// gpt-image-2-5-fal-api-docs.md).
|
|
157
|
+
//
|
|
158
|
+
// 🚨 IT WAS 10 UNTIL 2026-09-09, AND 10 WAS NEVER ANYBODY'S LIMIT. The
|
|
159
|
+
// comment here used to call it "the 10-reference ceiling ... unchanged by the
|
|
160
|
+
// version bump", which reads as a verified fal constraint and was not one —
|
|
161
|
+
// nobody had checked. Six reference slots were being given away, worst on
|
|
162
|
+
// Sunburst, whose entire reason for shipping is multi-reference edit work.
|
|
163
|
+
// Raised by Eric 2026-09-09 ("did we go completely full-on with all of the
|
|
164
|
+
// options on fal?").
|
|
165
|
+
//
|
|
166
|
+
// The five aspect ratios ARE a product choice; fal takes any custom size
|
|
167
|
+
// inside its own bounds (see GPT_IMAGE_25_SIZES in slate/src/shared/
|
|
168
|
+
// pricing.ts). Two SLATES caps sit outside this file and are not model
|
|
169
|
+
// limits either: the MCP's 4,000-character prompt against fal's 32,000, and
|
|
170
|
+
// image quantity, which is a fan-out and has no provider ceiling at all.
|
|
171
|
+
'gpt-image-2-5-flare': {
|
|
138
172
|
aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
|
|
139
|
-
maxRefImages:
|
|
173
|
+
maxRefImages: 16,
|
|
174
|
+
},
|
|
175
|
+
'gpt-image-2-5-sunburst': {
|
|
176
|
+
aspectRatios: ['1:1', '16:9', '9:16', '4:3', '3:4'],
|
|
177
|
+
maxRefImages: 16,
|
|
140
178
|
},
|
|
141
179
|
'flux-2-max': {
|
|
142
180
|
aspectRatios: FULL_ASPECT_RATIOS,
|
|
@@ -368,6 +406,17 @@ export const MODEL_CAPABILITIES = {
|
|
|
368
406
|
},
|
|
369
407
|
'minimax-h3-max': {
|
|
370
408
|
aspectRatios: MINIMAX_H3_ASPECT_RATIOS,
|
|
409
|
+
// 🚨 ZERO, DECLARED — not omitted. `minimax/h3-max/reference-to-video`
|
|
410
|
+
// returns 404, so there is no transport for a reference of any modality.
|
|
411
|
+
// Leaving this undeclared does NOT mean "none": `getMaxRefImages` falls
|
|
412
|
+
// back to `?? 3` for the ingredients mode, which handed this row three
|
|
413
|
+
// reference slots it cannot send. Measured 2026-09-09 — a pinned image on
|
|
414
|
+
// an h3-max generation was accepted by the composer, dropped in transit,
|
|
415
|
+
// and the model rendered the prompt text alone, returning a different
|
|
416
|
+
// person than the reference. Frames (start/end) remain the ONLY image
|
|
417
|
+
// transport on this seat.
|
|
418
|
+
maxIngredientImages: 0,
|
|
419
|
+
maxRefImages: 0,
|
|
371
420
|
// 480p/768p ONLY — fal's post-train of the open weights, and the 2K
|
|
372
421
|
// upscaler was never open-sourced. Declaring the shorter ladder here IS the
|
|
373
422
|
// whole Max-seat mechanism: `assertVideoCapabilities` refuses 2K/4K on this
|
|
@@ -146,12 +146,20 @@ export const MODEL_FACTS = [
|
|
|
146
146
|
notes: 'HERO-FRAME / typography PREMIUM image tier. NB2 is about 95% of Pro — escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene or deep multi-element reasoning must be perfect, and say why.',
|
|
147
147
|
},
|
|
148
148
|
{
|
|
149
|
-
id: 'gpt-image-2',
|
|
149
|
+
id: 'gpt-image-2-5-flare',
|
|
150
150
|
route: 'generate',
|
|
151
|
-
label: 'GPT Image 2',
|
|
151
|
+
label: 'GPT Image 2.5 Flare',
|
|
152
152
|
kind: 'image',
|
|
153
|
-
...caps('gpt-image-2'),
|
|
154
|
-
notes: '
|
|
153
|
+
...caps('gpt-image-2-5-flare'),
|
|
154
|
+
notes: 'THE FAST GPT IMAGE SEAT — OpenAI\'s small model, optimized for SPEED, quality COMPARABLE to GPT Image 2 (not better) at roughly half the latency. Route here when speed matters: drafts, exploration, volume. TEXT / DIAGRAM / PANEL work — character sheets, shot grids, text-bearing panels. When quality outranks speed, escalate to Sunburst. Own content filter, distinct from Gemini\'s. Killed by a head-to-head at the intended crop going the other way.',
|
|
155
|
+
},
|
|
156
|
+
{
|
|
157
|
+
id: 'gpt-image-2-5-sunburst',
|
|
158
|
+
route: 'generate',
|
|
159
|
+
label: 'GPT Image 2.5 Sunburst',
|
|
160
|
+
kind: 'image',
|
|
161
|
+
...caps('gpt-image-2-5-sunburst'),
|
|
162
|
+
notes: 'THE QUALITY GPT IMAGE SEAT — OpenAI\'s most capable image model, higher quality than GPT Image 2, same price as Flare, deliberately SLOWER. Route here whenever quality outranks speed: finals, hero frames, photoreal people, and multi-reference edits where every reference must survive into one frame — its widest lead. Not for drafts; you pay latency on every frame. Explore on Flare, finish on Sunburst.',
|
|
155
163
|
},
|
|
156
164
|
{
|
|
157
165
|
id: 'flux-2-max',
|
|
@@ -83,8 +83,14 @@ const SEEDANCE = {
|
|
|
83
83
|
},
|
|
84
84
|
{
|
|
85
85
|
heading: 'Images, clips and audio in ONE generation',
|
|
86
|
-
example: 'Marcus (image 1) performs the motion from video 1
|
|
87
|
-
note: 'Attaching a clip does NOT mean "edit this clip". A video or audio attachment is a REFERENCE, numbered in the rail exactly like an image, and it sits alongside your images in the same generation — the composer cites them as "image N", "video N", "audio N", in rail order, and shows you the exact sentence before you press Generate. Reorder the tiles to change what those numbers mean. To actually rewrite a clip, use Edit with AI instead — that is a different, deliberate choice.',
|
|
86
|
+
example: 'Marcus (image 1) performs the motion from video 1 and uses the voice timbre from audio 1.',
|
|
87
|
+
note: 'Attaching a clip does NOT mean "edit this clip". A video or audio attachment is a REFERENCE, numbered in the rail exactly like an image, and it sits alongside your images in the same generation — the composer cites them as "image N", "video N", "audio N", in rail order, and shows you the exact sentence before you press Generate. Reorder the tiles to change what those numbers mean. To actually rewrite a clip, use Edit with AI instead — that is a different, deliberate choice. EACH MODALITY IS NUMBERED SEPARATELY, from 1 — two images and one clip are "image 1", "image 2" and "video 1", never a single running count, so an audio reference is "audio 1" no matter how many images sit in front of it.',
|
|
88
|
+
critical: true,
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
heading: 'Give a character a voice',
|
|
92
|
+
example: 'Sarah (image 1) uses the voice timbre from audio 1. She says, "We open in ten minutes."',
|
|
93
|
+
note: 'An audio reference can mean five different things to Seedance — music, dialogue, voice, tone or timbre — so SAY WHICH. Name it as the voice timbre and the clip supplies the voice while your prompt supplies the words; leave it unroled and the model falls back to dialogue, re-transcribes the clip and speaks ITS words instead (a real take came back as "a map called Slates" for "an app called Slates"). Bind each speaker in a sentence rather than by attachment order — position carries nothing: "Images 1-2 are Character 1 and correspond to Audio 1; Images 3-4 are Character 2 and correspond to Audio 2." Verbatim from ByteDance. Audio-alone works on 2.5; on 2.0 pair it with at least one image or video.',
|
|
88
94
|
critical: true,
|
|
89
95
|
},
|
|
90
96
|
{
|
|
@@ -613,7 +619,7 @@ const MINIMAX_H3 = {
|
|
|
613
619
|
{
|
|
614
620
|
heading: 'Cite references by number',
|
|
615
621
|
example: 'Marcus (image 1) walks into the workshop (image 2)...',
|
|
616
|
-
note: `H3 takes references as typed slots and expects plain numbered prose — image 1, video 1, audio 1. Do not hand-write angle-bracket tags. ${PARTIALS['reference-tips-short']}`,
|
|
622
|
+
note: `H3 takes references as typed slots and expects plain numbered prose — image 1, video 1, audio 1, each modality counted separately from 1. fal's own prompt-field description: "Refer to reference assets by their modality and order in the reference lists: Image 1, Image 2, Video 1, Audio 1, and so on." Do not hand-write angle-bracket tags — MiniMax's guides use \`<Subject N>\` / \`<Audio N>\` as DOCUMENTATION notation and typing them puts literal brackets in the prompt. An audio reference binds as a voice timbre to a named speaker, the same primitive Seedance uses. ${PARTIALS['reference-tips-short']}`,
|
|
617
623
|
},
|
|
618
624
|
{
|
|
619
625
|
heading: 'Say how much of a reference survives',
|
|
@@ -107,6 +107,20 @@ export function isHexColorToken(sigil, token) {
|
|
|
107
107
|
}
|
|
108
108
|
export function composeReferences(rawPrompt, groups, opts = {}) {
|
|
109
109
|
// ── 1. Assign global numbers by walking the list in order ──
|
|
110
|
+
//
|
|
111
|
+
// 🚨 ONE COUNTER PER MODALITY, EACH STARTING AT 1 — never a single running
|
|
112
|
+
// count across the attachments. `audio 1` is the first AUDIO, however many
|
|
113
|
+
// images precede it. This is the vendors' scheme, not a convenience:
|
|
114
|
+
// BytePlus states it ("The numbering should correspond to the upload order of
|
|
115
|
+
// the assets, such as Image 1 / Video 1 / Audio 1") and PROVES it in a mixed
|
|
116
|
+
// example that numbers two audios 1 and 2 behind four images — a global
|
|
117
|
+
// counter would make them 5 and 6. fal says the same for MiniMax H3 in its
|
|
118
|
+
// own prompt-field description. Receipts and line refs:
|
|
119
|
+
// second-brain/business/projects/slates/research/model-prompting-research.md
|
|
120
|
+
// § 2026-09-09 Multimodal reference GRAMMAR.
|
|
121
|
+
//
|
|
122
|
+
// Collapsing these into one counter would renumber every citation the prompt
|
|
123
|
+
// makes, silently — the model would be told "audio 1" about an image.
|
|
110
124
|
let imageNum = opts.startImageNumber ?? 0;
|
|
111
125
|
let videoNum = opts.startVideoNumber ?? 0;
|
|
112
126
|
let audioNum = opts.startAudioNumber ?? 0;
|
|
@@ -60,8 +60,15 @@ export interface ShotParams {
|
|
|
60
60
|
imageResolution?: string;
|
|
61
61
|
videoResolution?: string;
|
|
62
62
|
quality?: string;
|
|
63
|
-
/**
|
|
64
|
-
|
|
63
|
+
/** GPT Image 2.5's tier. Always sent explicitly: fal's own default is
|
|
64
|
+
* `high`, which is the third of five rungs, not the top of two. */
|
|
65
|
+
gptQuality?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
|
|
66
|
+
/** GPT Image's alpha switch. `auto` is fal's default and ours; `transparent`
|
|
67
|
+
* asks for a real alpha channel rather than a painted backdrop. Costs
|
|
68
|
+
* nothing — fal prices this family on size × quality only, so it is NOT a
|
|
69
|
+
* cost-key segment. Named `gptBackground` because `background` already
|
|
70
|
+
* means "generate asynchronously" on every op that carries a Shot. */
|
|
71
|
+
gptBackground?: 'auto' | 'transparent' | 'opaque';
|
|
65
72
|
duration?: number;
|
|
66
73
|
imageQuantity?: number;
|
|
67
74
|
gridMode?: 'off' | '2x2' | '3x3';
|