@amaster.ai/pi-video-gen 0.1.2-beta.48
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +196 -0
- package/dist/compose.d.ts +43 -0
- package/dist/compose.d.ts.map +1 -0
- package/dist/compose.js +308 -0
- package/dist/compose.js.map +1 -0
- package/dist/config.d.ts +29 -0
- package/dist/config.d.ts.map +1 -0
- package/dist/config.js +305 -0
- package/dist/config.js.map +1 -0
- package/dist/errors.d.ts +50 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +99 -0
- package/dist/errors.js.map +1 -0
- package/dist/ffmpeg.d.ts +64 -0
- package/dist/ffmpeg.d.ts.map +1 -0
- package/dist/ffmpeg.js +292 -0
- package/dist/ffmpeg.js.map +1 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +837 -0
- package/dist/index.js.map +1 -0
- package/dist/jobs/store.d.ts +124 -0
- package/dist/jobs/store.d.ts.map +1 -0
- package/dist/jobs/store.js +386 -0
- package/dist/jobs/store.js.map +1 -0
- package/dist/providers/ark.d.ts +26 -0
- package/dist/providers/ark.d.ts.map +1 -0
- package/dist/providers/ark.js +178 -0
- package/dist/providers/ark.js.map +1 -0
- package/dist/providers/dashscope.d.ts +28 -0
- package/dist/providers/dashscope.d.ts.map +1 -0
- package/dist/providers/dashscope.js +200 -0
- package/dist/providers/dashscope.js.map +1 -0
- package/dist/providers/kling.d.ts +46 -0
- package/dist/providers/kling.d.ts.map +1 -0
- package/dist/providers/kling.js +245 -0
- package/dist/providers/kling.js.map +1 -0
- package/dist/providers/models.d.ts +16 -0
- package/dist/providers/models.d.ts.map +1 -0
- package/dist/providers/models.js +152 -0
- package/dist/providers/models.js.map +1 -0
- package/dist/providers/newapi.d.ts +3 -0
- package/dist/providers/newapi.d.ts.map +1 -0
- package/dist/providers/newapi.js +203 -0
- package/dist/providers/newapi.js.map +1 -0
- package/dist/providers/openrouter.d.ts +21 -0
- package/dist/providers/openrouter.d.ts.map +1 -0
- package/dist/providers/openrouter.js +169 -0
- package/dist/providers/openrouter.js.map +1 -0
- package/dist/providers/request.d.ts +4 -0
- package/dist/providers/request.d.ts.map +1 -0
- package/dist/providers/request.js +20 -0
- package/dist/providers/request.js.map +1 -0
- package/dist/providers/task.d.ts +53 -0
- package/dist/providers/task.d.ts.map +1 -0
- package/dist/providers/task.js +209 -0
- package/dist/providers/task.js.map +1 -0
- package/dist/render.d.ts +53 -0
- package/dist/render.d.ts.map +1 -0
- package/dist/render.js +480 -0
- package/dist/render.js.map +1 -0
- package/dist/text-layer.d.ts +18 -0
- package/dist/text-layer.d.ts.map +1 -0
- package/dist/text-layer.js +168 -0
- package/dist/text-layer.js.map +1 -0
- package/dist/timeline-render.d.ts +33 -0
- package/dist/timeline-render.d.ts.map +1 -0
- package/dist/timeline-render.js +1051 -0
- package/dist/timeline-render.js.map +1 -0
- package/dist/timeline.d.ts +64 -0
- package/dist/timeline.d.ts.map +1 -0
- package/dist/timeline.js +246 -0
- package/dist/timeline.js.map +1 -0
- package/dist/tts/edge-tts.d.ts +22 -0
- package/dist/tts/edge-tts.d.ts.map +1 -0
- package/dist/tts/edge-tts.js +55 -0
- package/dist/tts/edge-tts.js.map +1 -0
- package/dist/types.d.ts +161 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +13 -0
- package/dist/types.js.map +1 -0
- package/package.json +93 -0
- package/preview.png +0 -0
- package/skills/video-gen/SKILL.md +284 -0
package/dist/types.d.ts
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared types for pi-video-gen.
|
|
3
|
+
*
|
|
4
|
+
* The provider layer mirrors pi-image-gen's three-level split:
|
|
5
|
+
* wire format (`VideoApiStyle`) → model registry (`BuiltInVideoModel`) →
|
|
6
|
+
* resolved provider/model at runtime. Unlike image generation, vendor video
|
|
7
|
+
* APIs are async task APIs, so the adapter seam exposes the remote task
|
|
8
|
+
* lifecycle (`submit` / `inspect` / `downloadTo` / `cancel?`) instead of a
|
|
9
|
+
* single `generate()` — the remote task handle is what makes crash resume
|
|
10
|
+
* possible without double-billing.
|
|
11
|
+
*/
|
|
12
|
+
/** Wire format, NOT vendor — the same model via official or proxy endpoints differs only in baseUrl. */
|
|
13
|
+
export type VideoApiStyle = 'ark' | 'kling' | 'dashscope' | 'openrouter' | 'newapi';
|
|
14
|
+
export type VideoModelCapabilities = {
|
|
15
|
+
/** Max reference images a request may carry (0 = text-to-video only). */
|
|
16
|
+
maxReferenceImages: number;
|
|
17
|
+
/** Inclusive [min, max] clip duration in seconds. */
|
|
18
|
+
durations: [number, number];
|
|
19
|
+
resolutions: string[];
|
|
20
|
+
aspectRatios: string[];
|
|
21
|
+
/** Model generates synchronized speech/SFX/BGM natively (e.g. Seedance 2.0 `generate_audio`). */
|
|
22
|
+
nativeAudio: boolean;
|
|
23
|
+
/** First+last frame interpolation is actually honored by this endpoint. */
|
|
24
|
+
supportsFirstLastFrame: boolean;
|
|
25
|
+
};
|
|
26
|
+
export type BuiltInVideoModel = {
|
|
27
|
+
id: string;
|
|
28
|
+
aliases: string[];
|
|
29
|
+
provider: VideoApiStyle;
|
|
30
|
+
/** Remote model id sent to the provider (defaults to id). */
|
|
31
|
+
remoteId?: string;
|
|
32
|
+
capabilities: VideoModelCapabilities;
|
|
33
|
+
defaultResolution: string;
|
|
34
|
+
defaultAspectRatio: string;
|
|
35
|
+
defaultDurationSec: number;
|
|
36
|
+
};
|
|
37
|
+
export type ProviderSettings = {
|
|
38
|
+
apiKey?: string | undefined;
|
|
39
|
+
baseUrl?: string | undefined;
|
|
40
|
+
};
|
|
41
|
+
export type RateLimitSettings = {
|
|
42
|
+
maxRequestsPerMinute?: number | undefined;
|
|
43
|
+
maxRequestsPerDay?: number | undefined;
|
|
44
|
+
};
|
|
45
|
+
/** A user-defined video model routed through a custom (or built-in) provider. */
|
|
46
|
+
export type CustomVideoModel = {
|
|
47
|
+
/** Remote model id sent to the provider. */
|
|
48
|
+
id: string;
|
|
49
|
+
/** Optional alias the agent / user can refer to. */
|
|
50
|
+
alias?: string | undefined;
|
|
51
|
+
/** Optional display name. */
|
|
52
|
+
name?: string | undefined;
|
|
53
|
+
/**
|
|
54
|
+
* Capability declaration driving tool schema and preflight. Omitted fields
|
|
55
|
+
* fall back to conservative defaults (no audio, no last-frame, 720p, 16:9).
|
|
56
|
+
*/
|
|
57
|
+
capabilities?: VideoModelCapabilities | undefined;
|
|
58
|
+
defaultResolution?: string | undefined;
|
|
59
|
+
defaultAspectRatio?: string | undefined;
|
|
60
|
+
defaultDurationSec?: number | undefined;
|
|
61
|
+
};
|
|
62
|
+
/** A user-defined video-generation provider reusing a built-in wire format. */
|
|
63
|
+
export type CustomVideoProvider = {
|
|
64
|
+
/**
|
|
65
|
+
* Wire shape this provider speaks. Determines which adapter calls it.
|
|
66
|
+
* Values are video-generation API shapes (e.g. the Ark task API), NOT pi.dev
|
|
67
|
+
* LLM streaming formats.
|
|
68
|
+
*/
|
|
69
|
+
api: VideoApiStyle;
|
|
70
|
+
/**
|
|
71
|
+
* Override the API base URL. Optional for apis with a public default
|
|
72
|
+
* endpoint; REQUIRED for `newapi` (self-hosted relay — resolution fails
|
|
73
|
+
* without it).
|
|
74
|
+
*/
|
|
75
|
+
baseUrl?: string | undefined;
|
|
76
|
+
/** API key. User/agent-dir settings support `$ENV_VAR` and `${ENV_VAR}` syntax. */
|
|
77
|
+
apiKey?: string | undefined;
|
|
78
|
+
/** Optional display name. */
|
|
79
|
+
name?: string | undefined;
|
|
80
|
+
/** Models routed through this provider. */
|
|
81
|
+
models?: Array<string | CustomVideoModel>;
|
|
82
|
+
};
|
|
83
|
+
export type VideoGenSettings = {
|
|
84
|
+
/** Job root directory. Defaults to `<cwd>/.video-gen`. */
|
|
85
|
+
outputDir?: string;
|
|
86
|
+
/** Model id or alias resolved against the built-in registry. */
|
|
87
|
+
defaultModel?: string;
|
|
88
|
+
/** Sensitive: honored from global/agent-dir settings layers only. */
|
|
89
|
+
providers?: Partial<Record<VideoApiStyle, ProviderSettings>>;
|
|
90
|
+
/** Sensitive: honored from global/agent-dir settings layers only. */
|
|
91
|
+
ffmpegPath?: string;
|
|
92
|
+
rateLimit?: RateLimitSettings;
|
|
93
|
+
concurrency?: {
|
|
94
|
+
clips?: number;
|
|
95
|
+
};
|
|
96
|
+
/**
|
|
97
|
+
* User-defined custom providers keyed by provider name (mirrors
|
|
98
|
+
* pi-image-gen's customProviders). Sensitive: global/agent-dir layers only.
|
|
99
|
+
*/
|
|
100
|
+
customProviders?: Record<string, CustomVideoProvider>;
|
|
101
|
+
};
|
|
102
|
+
export type ResolvedProvider = {
|
|
103
|
+
style: VideoApiStyle;
|
|
104
|
+
apiKey?: string | undefined;
|
|
105
|
+
apiKeyPath?: string | undefined;
|
|
106
|
+
baseUrl: string;
|
|
107
|
+
};
|
|
108
|
+
export type ResolvedModel = {
|
|
109
|
+
entry: BuiltInVideoModel;
|
|
110
|
+
remoteId: string;
|
|
111
|
+
provider: ResolvedProvider;
|
|
112
|
+
};
|
|
113
|
+
export type GenerateVideoParams = {
|
|
114
|
+
prompt: string;
|
|
115
|
+
/** Stable orchestration identity for provider idempotency/recovery keys. */
|
|
116
|
+
requestId?: string | undefined;
|
|
117
|
+
firstFramePath?: string | undefined;
|
|
118
|
+
lastFramePath?: string | undefined;
|
|
119
|
+
/** Additional reference images (subject/style), roles mapped per provider. */
|
|
120
|
+
referenceImagePaths?: string[] | undefined;
|
|
121
|
+
durationSec?: number | undefined;
|
|
122
|
+
aspectRatio?: string | undefined;
|
|
123
|
+
resolution?: string | undefined;
|
|
124
|
+
/** Caller-side decision from capabilities.nativeAudio; adapter forwards it. */
|
|
125
|
+
generateAudio?: boolean | undefined;
|
|
126
|
+
};
|
|
127
|
+
export type RemoteTaskHandle = {
|
|
128
|
+
taskId: string;
|
|
129
|
+
/** ISO-8601 submission time. */
|
|
130
|
+
submittedAt: string;
|
|
131
|
+
/** Hash of model+prompt+frames+params, used to match a handle to its request on resume. */
|
|
132
|
+
requestFingerprint: string;
|
|
133
|
+
/** Adapter-specific resume data (e.g. Kling's task kind for the poll URL). */
|
|
134
|
+
meta?: Record<string, string> | undefined;
|
|
135
|
+
};
|
|
136
|
+
export type RemoteTaskStatus = {
|
|
137
|
+
phase: 'pending' | 'running';
|
|
138
|
+
} | {
|
|
139
|
+
phase: 'succeeded';
|
|
140
|
+
videoUrl: string;
|
|
141
|
+
} | {
|
|
142
|
+
phase: 'failed';
|
|
143
|
+
message: string;
|
|
144
|
+
};
|
|
145
|
+
export type VideoFileMeta = {
|
|
146
|
+
path: string;
|
|
147
|
+
bytes: number;
|
|
148
|
+
};
|
|
149
|
+
export type VideoProviderAdapter = {
|
|
150
|
+
/** Create the remote task. Caller persists the returned handle immediately. */
|
|
151
|
+
submit(provider: ResolvedProvider, remoteModelId: string, params: GenerateVideoParams, fetchImpl: typeof fetch, signal?: AbortSignal): Promise<RemoteTaskHandle>;
|
|
152
|
+
/** Query task state. Resume entry point: with a handle, never re-submit. */
|
|
153
|
+
inspect(provider: ResolvedProvider, handle: RemoteTaskHandle, fetchImpl: typeof fetch, signal?: AbortSignal): Promise<RemoteTaskStatus>;
|
|
154
|
+
/** Stream the finished video to destPath (temp file → limits → magic bytes → atomic rename). */
|
|
155
|
+
downloadTo(provider: ResolvedProvider, handle: RemoteTaskHandle, videoUrl: string, destPath: string, fetchImpl: typeof fetch, signal?: AbortSignal): Promise<VideoFileMeta>;
|
|
156
|
+
/** Only implemented when the vendor exposes a cancel API; absent ⇒ local stop is `polling_stopped`. */
|
|
157
|
+
cancel?(provider: ResolvedProvider, handle: RemoteTaskHandle, fetchImpl: typeof fetch, signal?: AbortSignal): Promise<{
|
|
158
|
+
cancelled: boolean;
|
|
159
|
+
}>;
|
|
160
|
+
};
|
|
161
|
+
//# sourceMappingURL=types.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,wGAAwG;AACxG,MAAM,MAAM,aAAa,GAAG,KAAK,GAAG,OAAO,GAAG,WAAW,GAAG,YAAY,GAAG,QAAQ,CAAC;AAEpF,MAAM,MAAM,sBAAsB,GAAG;IACnC,yEAAyE;IACzE,kBAAkB,EAAE,MAAM,CAAC;IAC3B,qDAAqD;IACrD,SAAS,EAAE,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC5B,WAAW,EAAE,MAAM,EAAE,CAAC;IACtB,YAAY,EAAE,MAAM,EAAE,CAAC;IACvB,iGAAiG;IACjG,WAAW,EAAE,OAAO,CAAC;IACrB,2EAA2E;IAC3E,sBAAsB,EAAE,OAAO,CAAC;CACjC,CAAC;AAEF,MAAM,MAAM,iBAAiB,GAAG;IAC9B,EAAE,EAAE,MAAM,CAAC;IACX,OAAO,EAAE,MAAM,EAAE,CAAC;IAClB,QAAQ,EAAE,aAAa,CAAC;IACxB,6DAA6D;IAC7D,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,YAAY,EAAE,sBAAsB,CAAC;IACrC,iBAAiB,EAAE,MAAM,CAAC;IAC1B,kBAAkB,EAAE,MAAM,CAAC;IAC3B,kBAAkB,EAAE,MAAM,CAAC;CAC5B,CAAC;AAEF,MAAM,MAAM,gBAAgB,GAAG;IAC7B,MAAM,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC5B,OAAO,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;CAC9B,CAAC;AAEF,MAAM,MAAM,iBAAiB,GAAG;IAC9B,oBAAoB,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC1C,iBAAiB,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;CACxC,CAAC;AAEF,iFAAiF;AACjF,MAAM,MAAM,gBAAgB,GAAG;IAC7B,4CAA4C;IAC5C,EAAE,EAAE,MAAM,CAAC;IACX,oDAAoD;IACpD,KAAK,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC3B,6BAA6B;IAC7B,IAAI,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC1B;;;OAGG;IACH,YAAY,CAAC,EAAE,sBAAsB,GAAG,SAAS,CAAC;IAClD,iBAAiB,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACvC,kBAAkB,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACxC,kBAAkB,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;CACzC,CAAC;AAEF,+EAA+E;AAC/E,MAAM,MAAM,mBAAmB,GAAG;IAChC;;;;OAIG;IACH,GAAG,EAAE,aAAa,CAAC;IACnB;;;;OAIG;IACH,OAAO,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC7B,mFAAmF;IACnF,MAAM,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC5B,6BAA6B;IAC7B,IAAI,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC1B,2CAA2C;IAC3C,MAAM,CAAC,EAAE,KAAK,CAAC,MAAM,GAAG,gBAAgB,CAAC,CAAC;CAC3C,CAAC;AAEF,MAAM,MAAM,gBAAgB,GAAG;IAC7B,0DAA0D;IAC1D,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,gEAAgE;IAChE,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,qEAAqE;IACrE,SAAS,CAAC,EAAE,OAAO,CAAC,MAAM,CAAC,aAAa,EAAE,gBAAgB,CAAC,CAAC,CAAC;IAC7D,qEAAqE;IACrE,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,SAAS,CAAC,EAAE,iBAAiB,CAAC;IAC9B,WAAW,CAAC,EAAE;QAAE,KAAK,CAAC,EAAE,MAAM,CAAA;KAAE,CAAC;IACjC;;;OAGG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,mBAAmB,CAAC,CAAC;CACvD,CAAC;AAEF,MAAM,MAAM,gBAAgB,GAAG;IAC7B,KAAK,EAAE,aAAa,CAAC;IACrB,MAAM,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC5B,UAAU,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAChC,OAAO,EAAE,MAAM,CAAC;CACjB,CAAC;AAEF,MAAM,MAAM,aAAa,GAAG;IAC1B,KAAK,EAAE,iBAAiB,CAAC;IACzB,QAAQ,EAAE,MAAM,CAAC;IACjB,QAAQ,EAAE,gBAAgB,CAAC;CAC5B,CAAC;AAEF,MAAM,MAAM,mBAAmB,GAAG;IAChC,MAAM,EAAE,MAAM,CAAC;IACf,4EAA4E;IAC5E,SAAS,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC/B,cAAc,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACpC,aAAa,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACnC,8EAA8E;IAC9E,mBAAmB,CAAC,EAAE,MAAM,EAAE,GAAG,SAAS,CAAC;IAC3C,WAAW,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACjC,WAAW,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IACjC,UAAU,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAChC,+EAA+E;IAC/E,aAAa,CAAC,EAAE,OAAO,GAAG,SAAS,CAAC;CACrC,CAAC;AAEF,MAAM,MAAM,gBAAgB,GAAG;IAC7B,MAAM,EAAE,MAAM,CAAC;IACf,gCAAgC;IAChC,WAAW,EAAE,MAAM,CAAC;IACpB,2FAA2F;IAC3F,kBAAkB,EAAE,MAAM,CAAC;IAC3B,8EAA8E;IAC9E,IAAI,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,GAAG,SAAS,CAAC;CAC3C,CAAC;AAEF,MAAM,MAAM,gBAAgB,GACxB;IAAE,KAAK,EAAE,SAAS,GAAG,SAAS,CAAA;CAAE,GAChC;IAAE,KAAK,EAAE,WAAW,CAAC;IAAC,QAAQ,EAAE,MAAM,CAAA;CAAE,GACxC;IAAE,KAAK,EAAE,QAAQ,CAAC;IAAC,OAAO,EAAE,MAAM,CAAA;CAAE,CAAC;AAEzC,MAAM,MAAM,aAAa,GAAG;IAC1B,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,MAAM,CAAC;CACf,CAAC;AAEF,MAAM,MAAM,oBAAoB,GAAG;IACjC,+EAA+E;IAC/E,MAAM,CACJ,QAAQ,EAAE,gBAAgB,EAC1B,aAAa,EAAE,MAAM,EACrB,MAAM,EAAE,mBAAmB,EAC3B,SAAS,EAAE,OAAO,KAAK,EACvB,MAAM,CAAC,EAAE,WAAW,GACnB,OAAO,CAAC,gBAAgB,CAAC,CAAC;IAE7B,4EAA4E;IAC5E,OAAO,CACL,QAAQ,EAAE,gBAAgB,EAC1B,MAAM,EAAE,gBAAgB,EACxB,SAAS,EAAE,OAAO,KAAK,EACvB,MAAM,CAAC,EAAE,WAAW,GACnB,OAAO,CAAC,gBAAgB,CAAC,CAAC;IAE7B,gGAAgG;IAChG,UAAU,CACR,QAAQ,EAAE,gBAAgB,EAC1B,MAAM,EAAE,gBAAgB,EACxB,QAAQ,EAAE,MAAM,EAChB,QAAQ,EAAE,MAAM,EAChB,SAAS,EAAE,OAAO,KAAK,EACvB,MAAM,CAAC,EAAE,WAAW,GACnB,OAAO,CAAC,aAAa,CAAC,CAAC;IAE1B,uGAAuG;IACvG,MAAM,CAAC,CACL,QAAQ,EAAE,gBAAgB,EAC1B,MAAM,EAAE,gBAAgB,EACxB,SAAS,EAAE,OAAO,KAAK,EACvB,MAAM,CAAC,EAAE,WAAW,GACnB,OAAO,CAAC;QAAE,SAAS,EAAE,OAAO,CAAA;KAAE,CAAC,CAAC;CACpC,CAAC"}
|
package/dist/types.js
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared types for pi-video-gen.
|
|
3
|
+
*
|
|
4
|
+
* The provider layer mirrors pi-image-gen's three-level split:
|
|
5
|
+
* wire format (`VideoApiStyle`) → model registry (`BuiltInVideoModel`) →
|
|
6
|
+
* resolved provider/model at runtime. Unlike image generation, vendor video
|
|
7
|
+
* APIs are async task APIs, so the adapter seam exposes the remote task
|
|
8
|
+
* lifecycle (`submit` / `inspect` / `downloadTo` / `cancel?`) instead of a
|
|
9
|
+
* single `generate()` — the remote task handle is what makes crash resume
|
|
10
|
+
* possible without double-billing.
|
|
11
|
+
*/
|
|
12
|
+
export {};
|
|
13
|
+
//# sourceMappingURL=types.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.js","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG"}
|
package/package.json
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@amaster.ai/pi-video-gen",
|
|
3
|
+
"version": "0.1.2-beta.48",
|
|
4
|
+
"description": "Pi extension for AI video generation plus local video composition: lossless clip concat and mixed image/video timelines with overlays, TTS, soft or burned subtitles, source audio, BGM, and bundled LGPL/GPL FFmpeg runtimes.",
|
|
5
|
+
"keywords": [
|
|
6
|
+
"pi-package",
|
|
7
|
+
"pi",
|
|
8
|
+
"extension",
|
|
9
|
+
"video-generation",
|
|
10
|
+
"seedance",
|
|
11
|
+
"volcengine",
|
|
12
|
+
"ark",
|
|
13
|
+
"kling",
|
|
14
|
+
"agentic-video"
|
|
15
|
+
],
|
|
16
|
+
"license": "Apache-2.0",
|
|
17
|
+
"type": "module",
|
|
18
|
+
"sideEffects": false,
|
|
19
|
+
"main": "./dist/index.js",
|
|
20
|
+
"types": "./dist/index.d.ts",
|
|
21
|
+
"exports": {
|
|
22
|
+
".": {
|
|
23
|
+
"types": "./dist/index.d.ts",
|
|
24
|
+
"default": "./dist/index.js"
|
|
25
|
+
},
|
|
26
|
+
"./package.json": {
|
|
27
|
+
"default": "./package.json"
|
|
28
|
+
}
|
|
29
|
+
},
|
|
30
|
+
"files": [
|
|
31
|
+
"dist",
|
|
32
|
+
"skills",
|
|
33
|
+
"README.md",
|
|
34
|
+
"preview.png"
|
|
35
|
+
],
|
|
36
|
+
"pi": {
|
|
37
|
+
"extensions": [
|
|
38
|
+
"./dist/index.js"
|
|
39
|
+
],
|
|
40
|
+
"skills": [
|
|
41
|
+
"./skills"
|
|
42
|
+
],
|
|
43
|
+
"image": "https://raw.githubusercontent.com/TGYD-helige/pi/master/packages/pi-video-gen/preview.png"
|
|
44
|
+
},
|
|
45
|
+
"publishConfig": {
|
|
46
|
+
"access": "public"
|
|
47
|
+
},
|
|
48
|
+
"repository": {
|
|
49
|
+
"type": "git",
|
|
50
|
+
"url": "https://github.com/TGYD-helige/pi.git",
|
|
51
|
+
"directory": "packages/pi-video-gen"
|
|
52
|
+
},
|
|
53
|
+
"dependencies": {
|
|
54
|
+
"sharp": "^0.34.0",
|
|
55
|
+
"msedge-tts": "^2.0.7",
|
|
56
|
+
"@amaster.ai/pi-shared": "0.1.2-beta.48"
|
|
57
|
+
},
|
|
58
|
+
"optionalDependencies": {
|
|
59
|
+
"@amaster.ai/pi-video-gen-ffmpeg-darwin-x64": "0.1.2-beta.48",
|
|
60
|
+
"@amaster.ai/pi-video-gen-ffmpeg-linux-arm64": "0.1.2-beta.48",
|
|
61
|
+
"@amaster.ai/pi-video-gen-ffmpeg-darwin-arm64": "0.1.2-beta.48",
|
|
62
|
+
"@amaster.ai/pi-video-gen-ffmpeg-linux-x64": "0.1.2-beta.48",
|
|
63
|
+
"@amaster.ai/pi-video-gen-ffmpeg-win32-x64": "0.1.2-beta.48"
|
|
64
|
+
},
|
|
65
|
+
"peerDependencies": {
|
|
66
|
+
"@earendil-works/pi-ai": ">=0.80.10",
|
|
67
|
+
"@earendil-works/pi-coding-agent": ">=0.74.0",
|
|
68
|
+
"typebox": "*"
|
|
69
|
+
},
|
|
70
|
+
"peerDependenciesMeta": {
|
|
71
|
+
"@earendil-works/pi-ai": {
|
|
72
|
+
"optional": true
|
|
73
|
+
},
|
|
74
|
+
"@earendil-works/pi-coding-agent": {
|
|
75
|
+
"optional": true
|
|
76
|
+
},
|
|
77
|
+
"typebox": {
|
|
78
|
+
"optional": true
|
|
79
|
+
}
|
|
80
|
+
},
|
|
81
|
+
"devDependencies": {
|
|
82
|
+
"@earendil-works/pi-ai": "0.80.10",
|
|
83
|
+
"@earendil-works/pi-coding-agent": "0.80.10",
|
|
84
|
+
"typebox": "*",
|
|
85
|
+
"vitest": "^4.0.0",
|
|
86
|
+
"ffmpeg-static": "^5.2.0"
|
|
87
|
+
},
|
|
88
|
+
"scripts": {
|
|
89
|
+
"build": "tsc -b",
|
|
90
|
+
"typecheck": "tsc -b --pretty false",
|
|
91
|
+
"test": "vitest run src"
|
|
92
|
+
}
|
|
93
|
+
}
|
package/preview.png
ADDED
|
Binary file
|
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: video-gen
|
|
3
|
+
description: "Video creation and local composition: join existing clips or render mixed image/video timelines with video_compose, generate one AI clip with video_generate, or make a multi-shot AI film with video_render. Use when the deliverable is a video. Do NOT use for still images (use pi-image-gen directly)."
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Video generation
|
|
7
|
+
|
|
8
|
+
This skill orchestrates four published flows:
|
|
9
|
+
|
|
10
|
+
- **C0 local concat** — `video_compose` joins compatible existing MP4 clips.
|
|
11
|
+
- **Timeline local render** — `video_compose` turns images/screenshots and
|
|
12
|
+
existing video clips into a video with overlays, TTS, motion, transitions,
|
|
13
|
+
soft or burned subtitles, source audio, and optional BGM.
|
|
14
|
+
- **Single AI clip** — one paid `video_generate` call.
|
|
15
|
+
- **Shot-book AI film** — author a shot book, generate frames with
|
|
16
|
+
`image_generate`, then ONE paid `video_render` call renders and stitches.
|
|
17
|
+
|
|
18
|
+
The AI video model is fixed by `pi-video-gen.defaultModel`; generated images use
|
|
19
|
+
pi-image-gen's active model.
|
|
20
|
+
|
|
21
|
+
## A. Workflow rules
|
|
22
|
+
|
|
23
|
+
0. **Route first.** Choose the right flow before anything:
|
|
24
|
+
|
|
25
|
+
| User goal | Flow |
|
|
26
|
+
|---|---|
|
|
27
|
+
| Existing local mp4 clips to join | `video_compose` (C0 — lossless, local, no paid models) |
|
|
28
|
+
| One AI-generated moving shot | `video_generate` |
|
|
29
|
+
| Multi-shot film with keyframes | `video_render` |
|
|
30
|
+
| Promo/explainer from images, screenshots & clips | `video_compose` (TimelineSpec — mixed media + overlays + TTS + transitions, local render, near-zero cost) |
|
|
31
|
+
|
|
32
|
+
1. **Local flows stop here.** For C0 follow §A0; for Timeline follow §A1. Do not
|
|
33
|
+
run the AI preflight, shot-book steps, or paid confirmation gates below.
|
|
34
|
+
Timeline only needs `image_generate` when its source images do not already
|
|
35
|
+
exist.
|
|
36
|
+
2. **AI preflight only.** For `video_generate` or `video_render`, call
|
|
37
|
+
`video_capabilities` and respect the active model's duration range and audio
|
|
38
|
+
support. Confirm `image_generate` is available only when source frames need
|
|
39
|
+
to be generated (`/video-gen doctor` checks; config health is
|
|
40
|
+
`/image-gen list`).
|
|
41
|
+
3. **Pick the AI flow.** A vague idea or a script that needs multiple shots →
|
|
42
|
+
shot-book flow. One moving shot → `video_generate`. A still → pi-image-gen.
|
|
43
|
+
4. **Write the shot book in conversation** (schema in §B). If the user only has a
|
|
44
|
+
vague idea, first be the screenwriter: three-act structure, filmable actions
|
|
45
|
+
("show, don't tell"), concrete visual detail. Iterate with the user in chat.
|
|
46
|
+
5. **Confirmation gate 1 (shot-book only).** Show the shot-book summary — shot count,
|
|
47
|
+
character list, estimated image calls (~2N+3C) and video calls (N) — and get an
|
|
48
|
+
explicit go-ahead. **Default small: 1 scene, 3–5 shots** unless the user asks
|
|
49
|
+
for more.
|
|
50
|
+
6. **Image stage (shot-book only, via `image_generate`, per §C).** Character portraits →
|
|
51
|
+
per-shot first frame (and last frame when needed). Show each batch to the user.
|
|
52
|
+
7. **Paid confirmation.** Before `video_generate`, confirm its one paid call.
|
|
53
|
+
For a shot book, once frames are ready, state "about to make N paid
|
|
54
|
+
video calls" and get an explicit render order. Then assemble the render spec
|
|
55
|
+
and call `video_render` ONCE.
|
|
56
|
+
8. **Cost honesty.** AI video calls are paid and take minutes each. Never state
|
|
57
|
+
amounts (prices change); state call counts and durations.
|
|
58
|
+
9. **Revisions.** The render spec is immutable per job directory. Text-stage
|
|
59
|
+
revisions happen in chat (regenerate frames as needed); a revised film goes in
|
|
60
|
+
a NEW job directory. NEVER suggest "delete shots/<id>/ and rerender" — that
|
|
61
|
+
breaks downstream dependencies. Rerunning the SAME spec path resumes an
|
|
62
|
+
interrupted job (finished shots don't re-bill).
|
|
63
|
+
10. **Degradation negotiation.** If `video_render` preflight fails (e.g. last
|
|
64
|
+
frame unsupported), present the options (switch model / edit spec /
|
|
65
|
+
`allowDegradations`) and let the user choose. Never degrade silently. When the
|
|
66
|
+
model's `nativeAudio` is false, don't write audio cues into video prompts
|
|
67
|
+
unless the user accepted silence.
|
|
68
|
+
11. **Cancellation honesty.** Interrupting stops local polling only — remote
|
|
69
|
+
tasks may keep running and billable (Ark cancellation is unverified). Say so.
|
|
70
|
+
|
|
71
|
+
## A0. C0 — composing existing clips (`video_compose`)
|
|
72
|
+
|
|
73
|
+
1. **Tell the user first**: clip count, order, output location
|
|
74
|
+
(`<jobDir>/final_video.mp4`), `mode: "copy"`. This is LOCAL compute — do
|
|
75
|
+
not use the paid-model confirmation script for it.
|
|
76
|
+
2. Write `<jobDir>/compose-input.json`
|
|
77
|
+
(`{"clips":[{"id":"c1","path":"/abs/a.mp4"},…],"output":{"mode":"copy"}}`)
|
|
78
|
+
under the video-gen output dir, then call `video_compose` ONCE. Keep source
|
|
79
|
+
clips outside `<jobDir>/clips/`; that directory and `final_video.mp4` are
|
|
80
|
+
reserved pipeline outputs, and a fresh job refuses either conflict.
|
|
81
|
+
3. **Only promise** lossless concat of compatible MP4s (C0). **Never promise**
|
|
82
|
+
trimming, transitions, overlays, subtitles, TTS, BGM, or re-encoding **for
|
|
83
|
+
the C0 path** — those live in the Timeline path (A1 below), not here; do
|
|
84
|
+
not hint at them for `compose-input.json`.
|
|
85
|
+
4. On any ordered stream incompatibility across all tracks
|
|
86
|
+
(codec/resolution/fps/timebase/pix_fmt/sample-rate/audio layout), hand the
|
|
87
|
+
exact ffprobe differences back to the user/agent:
|
|
88
|
+
re-encode the odd clips first. NEVER silently transcode, and NEVER fall
|
|
89
|
+
back to `video_generate`/`video_render` as a workaround.
|
|
90
|
+
5. Interrupted? Rerun the SAME path (fingerprint-verified resume / cached).
|
|
91
|
+
Changed clips or order? NEW job directory. A completed final video is
|
|
92
|
+
hash-bound; if it is missing or changed, restore the exact artifact or start
|
|
93
|
+
a NEW job.
|
|
94
|
+
|
|
95
|
+
## A1. Timeline compose (`video_compose` with `timeline-input.json`)
|
|
96
|
+
|
|
97
|
+
Use for promos/explainers from still images and existing clips. Costs ~0
|
|
98
|
+
(Edge TTS is free, render is local) — prefer it over AI video for this job type.
|
|
99
|
+
|
|
100
|
+
1. **Collect existing images/screenshots/clips first**, and use `image_generate`
|
|
101
|
+
only for missing visual material. Keep all source media outside the job
|
|
102
|
+
directory, then author `<jobDir>/timeline-input.json`.
|
|
103
|
+
`assets/`, `overlays/`, `audio/`, `segments/`, `qc/`, generated tracks,
|
|
104
|
+
subtitles, and `final_video.mp4` are reserved pipeline outputs; a fresh job
|
|
105
|
+
refuses any conflicts rather than deleting them.
|
|
106
|
+
```jsonc
|
|
107
|
+
{
|
|
108
|
+
"title": "产品宣传片",
|
|
109
|
+
"output": { "resolution": "1920x1080", "fps": 25, "codec": "h264" },
|
|
110
|
+
"voice": "edge-tts:zh-CN-YunyangNeural", // default; free, no key
|
|
111
|
+
"ttsFailureMode": "fail", // or "silent-subtitles" only after the user accepts that degradation
|
|
112
|
+
"subtitles": { "mode": "burn", "fontSize": 36,
|
|
113
|
+
"textColor": "#ffffff", "backgroundColor": "#000000", "backgroundOpacity": 0.55 },
|
|
114
|
+
"segments": [
|
|
115
|
+
{
|
|
116
|
+
"id": "intro",
|
|
117
|
+
"image": "/abs/frame-1.png",
|
|
118
|
+
"durationSec": 5, // image only: may be "auto" from narration
|
|
119
|
+
"motion": "kenburns-in", // image only
|
|
120
|
+
"transitionTo": { "type": "xfade", "style": "fade", "durationSec": 0.8 },
|
|
121
|
+
"overlay": { "title": "标题", "subtitle": "副标题", "position": "bottom-left" },
|
|
122
|
+
"narration": "这一段的中文旁白文本"
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
"id": "demo",
|
|
126
|
+
"video": "/abs/demo.mp4",
|
|
127
|
+
"trimStartSec": 2.5,
|
|
128
|
+
"durationSec": 6, // video always uses a numeric duration
|
|
129
|
+
"fit": "contain", // contain | cover
|
|
130
|
+
"sourceAudio": { "muted": false, "volume": 0.25 }
|
|
131
|
+
}
|
|
132
|
+
]
|
|
133
|
+
}
|
|
134
|
+
```
|
|
135
|
+
2. **Chinese text NEVER comes from an image model** — titles/subtitles go in
|
|
136
|
+
`overlay` and are rendered locally via SVG (no garbled CJK).
|
|
137
|
+
3. **Cost confirmation is unnecessary** (local compute), but still show the
|
|
138
|
+
segment count and total planned duration before calling `video_compose`.
|
|
139
|
+
4. Every segment contains exactly one of `image` or `video`. Video segments
|
|
140
|
+
are normalized to the output resolution/fps, may be trimmed/scaled, and
|
|
141
|
+
mix their source audio with narration before optional BGM. Video source
|
|
142
|
+
audio without a stream degrades to silence; `sourceAudio.muted: true` or
|
|
143
|
+
`volume: 0` disables it. A video's numeric `durationSec` is its fixed trim
|
|
144
|
+
window; narration that does not fit is rejected instead of extending it.
|
|
145
|
+
5. Narration uses Edge TTS (free). Measured audio duration drives image
|
|
146
|
+
`durationSec: "auto"`; subtitles use each segment's actual video timing.
|
|
147
|
+
`subtitles.mode` defaults to `"soft"` (`mov_text`); `"burn"` renders the
|
|
148
|
+
configured font/color/background directly into each narrated segment.
|
|
149
|
+
TTS failures stop the job by default. Use `ttsFailureMode:
|
|
150
|
+
"silent-subtitles"` only as an explicit degradation choice; it keeps the
|
|
151
|
+
subtitle track and fills that segment with silence. Once accepted, that
|
|
152
|
+
degradation is cached for the immutable job; create a NEW job to retry
|
|
153
|
+
real narration.
|
|
154
|
+
6. On completion, review the QC frames in `<jobDir>/qc/` yourself (Read the
|
|
155
|
+
PNGs) before showing the result — flipped/overlapping text only shows up
|
|
156
|
+
visually. Soft `mov_text` subtitles are not burned into those PNGs; the
|
|
157
|
+
pipeline separately verifies that the subtitle stream exists and that the
|
|
158
|
+
SRT cues match the resolved segment timeline.
|
|
159
|
+
7. The spec is immutable per job: rerunning the same path resumes only
|
|
160
|
+
regular job-local artifacts whose manifest hashes still match; changes
|
|
161
|
+
require a NEW job directory. A committed artifact that is missing or
|
|
162
|
+
changed is rejected rather than regenerated underneath cached downstream
|
|
163
|
+
outputs. A completed manifest with any missing artifact hash is rejected;
|
|
164
|
+
an interrupted manifest invalidates unverified downstream hashes before
|
|
165
|
+
rebuilding an uncommitted upstream artifact.
|
|
166
|
+
|
|
167
|
+
## B. Shot book (VideoProject) — authoring reference
|
|
168
|
+
|
|
169
|
+
Author as JSON in conversation; save to `<jobDir>/project.json` for the record.
|
|
170
|
+
|
|
171
|
+
```jsonc
|
|
172
|
+
{
|
|
173
|
+
"title": "...", "style": "Cartoon",
|
|
174
|
+
"characters": [{ "id": "alice", "visible": true,
|
|
175
|
+
"appearance": "long blonde hair, blue eyes, slender", // static features
|
|
176
|
+
"outfit": "red scarf, black leather jacket" }], // dynamic features
|
|
177
|
+
"shots": [{
|
|
178
|
+
"id": "s1",
|
|
179
|
+
"intent": "Wide shot, rainy alley. <Alice> enters from the left, stops under the streetlamp…",
|
|
180
|
+
"firstFrame": "…pure static description of the FIRST frame…",
|
|
181
|
+
"lastFrame": "…(optional) pure static description of the LAST frame…",
|
|
182
|
+
"motion": "Static camera. A woman with long blonde hair and a red scarf walks in from the left…",
|
|
183
|
+
"audio": "[Sound Effect] rain, distant traffic. [Speaker] Alice (soft): \"We're here.\"",
|
|
184
|
+
"visibleCharacters": ["alice"],
|
|
185
|
+
"durationSec": 5,
|
|
186
|
+
"continuityGroup": "alley",
|
|
187
|
+
"startFrameFromShotId": "s0", // optional: this shot's frame builds on s0's frame
|
|
188
|
+
"continuityNote": "In s0's frame Alice faces away; front view missing"
|
|
189
|
+
}]
|
|
190
|
+
}
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
Field rules:
|
|
194
|
+
|
|
195
|
+
- **Every shot needs a narrative purpose** (establish / emotion / reaction). First
|
|
196
|
+
shot: widest view of the scene. Close-ups for emotion, wide shots for context.
|
|
197
|
+
- **At most one dialogue line per shot.** Character names in `intent` are wrapped
|
|
198
|
+
in angle brackets: `<Alice>`.
|
|
199
|
+
- **firstFrame / lastFrame are pure static snapshots** — no ongoing actions
|
|
200
|
+
("he is sitting, leaning forward", NOT "he is about to stand"). Include shot
|
|
201
|
+
size, angle, composition, who is where and facing which way.
|
|
202
|
+
- **motion = camera movement + in-frame movement**, named separately. Refer to
|
|
203
|
+
characters by visible traits ("the woman in the red scarf"), never by name.
|
|
204
|
+
- **lastFrame needed when**: composition/focus changes drastically, a character
|
|
205
|
+
enters or turns to face camera, a major reveal happens. Otherwise omit it.
|
|
206
|
+
- **Few camera positions.** Default: one `continuityGroup` for everything. New
|
|
207
|
+
group only when shot size/angle/focus differs significantly.
|
|
208
|
+
- **continuityGroup** = shots sharing a space/base image; **startFrameFromShotId**
|
|
209
|
+
pins a specific parent frame for composition; **continuityNote** says what the
|
|
210
|
+
parent frame lacks (the frame prompt must then keep the background and replace
|
|
211
|
+
those elements). Self-check: parent shot EXISTS, comes EARLIER, same
|
|
212
|
+
continuityGroup, no cycles.
|
|
213
|
+
- **audio** uses `[Sound Effect] …` / `[Speaker] Name (Emotion): "line"` format.
|
|
214
|
+
- **durationSec and all capability values come from `video_capabilities`** —
|
|
215
|
+
never from memory or this document. Durations, resolutions, ratios, audio and
|
|
216
|
+
frame support differ per model and change over time.
|
|
217
|
+
- **Behavioral quirks worth knowing** (still verify with `video_capabilities`):
|
|
218
|
+
some models have no native audio (omit audio cues or the render is silent);
|
|
219
|
+
some cannot do last-frame interpolation (never pass lastFrame to them);
|
|
220
|
+
HappyHorse takes a first frame OR reference images in one call, not both —
|
|
221
|
+
cite references in the prompt as `[Image 1]`, `[Image 2]`, …
|
|
222
|
+
|
|
223
|
+
## C. Image operation manual (via `image_generate`)
|
|
224
|
+
|
|
225
|
+
Generic `image_generate` usage (params, sizes, `n`, edit labeling) follows the
|
|
226
|
+
**pi-image-gen skill** — it is the single authority; do not deviate. Two
|
|
227
|
+
video-specific handoff rules:
|
|
228
|
+
|
|
229
|
+
- **Never assume a saved filename**: the actual extension follows the MIME type
|
|
230
|
+
and collisions get `-v2`. **The returned absolute path is the only truth** —
|
|
231
|
+
record it immediately in `assets.json` (see below) and reference it in the
|
|
232
|
+
render spec.
|
|
233
|
+
- `assets.json` in the job dir: `{ "assets": { "<shotId>/<part>": { "sourcePath": "…" } } }`
|
|
234
|
+
mapping semantic assets (e.g. `s1/firstFrame`, `alice/front`) to real paths.
|
|
235
|
+
|
|
236
|
+
**Character portraits (3 views per visible character)**:
|
|
237
|
+
|
|
238
|
+
- front (text-to-image): `Generate a full-body, front-view portrait of character {identifier} based on the following description, with a pure white background. Use a wide 16:9 landscape canvas, not a vertical portrait canvas. The character should be centered in the image, occupying the middle of the wide frame with enough horizontal empty space. Gazing straight ahead. Standing with arms relaxed at sides. Natural expression. Features: {appearance}; {outfit}. Style: {style}`
|
|
239
|
+
- side (edit with front as reference): `Generate a full-body, side-view portrait of character {identifier} based on the provided front-view portrait, with a pure white background. Use a wide 16:9 landscape canvas, not a vertical portrait canvas. The character should be centered in the image, occupying the middle of the wide frame with enough horizontal empty space. Facing left. Standing with arms relaxed at sides.`
|
|
240
|
+
- back (edit with front as reference): `Generate a full-body, back-view portrait of character {identifier} based on the provided front-view portrait, with a pure white background. Use a wide 16:9 landscape canvas, not a vertical portrait canvas. The character should be centered in the image, occupying the middle of the wide frame with enough horizontal empty space. No facial features should be visible.`
|
|
241
|
+
|
|
242
|
+
If side/back fails after one retry, reuse front. Characters with `visible: false`
|
|
243
|
+
get no portraits.
|
|
244
|
+
|
|
245
|
+
**Reference selection for frames**:
|
|
246
|
+
candidates = portraits of visible characters (ONE view each, chosen by facing) +
|
|
247
|
+
continuity frames. Pick a SMALL set of the most relevant ones — same
|
|
248
|
+
camera/group first, most recent frames first, drop redundant near-duplicates,
|
|
249
|
+
prefer the portrait when a character newly appears. How many images a call
|
|
250
|
+
accepts is pi-image-gen's authority (its skill/tool description), not this
|
|
251
|
+
document's.
|
|
252
|
+
|
|
253
|
+
**Frame prompt assembly**: prefix each reference image with its role, then the
|
|
254
|
+
frame description mapping elements to images:
|
|
255
|
+
|
|
256
|
+
```
|
|
257
|
+
Image 0: A front view portrait of Alice.
|
|
258
|
+
Image 1: [alley] Wide shot of the rainy alley from shot s1.
|
|
259
|
+
Create an image based on the following description: <firstFrame text>. The alley
|
|
260
|
+
background should reference Image 1; Alice's appearance should reference Image 0.
|
|
261
|
+
```
|
|
262
|
+
|
|
263
|
+
## D. Assemble the render spec and render
|
|
264
|
+
|
|
265
|
+
Write `<outputDir>/<jobId>/render-input.json` (jobId: letters/digits/dash/underscore):
|
|
266
|
+
|
|
267
|
+
```jsonc
|
|
268
|
+
{
|
|
269
|
+
"title": "…", "aspectRatio": "16:9",
|
|
270
|
+
"shots": [{
|
|
271
|
+
"id": "s1",
|
|
272
|
+
"videoPrompt": "<motion> + <audio cues>",
|
|
273
|
+
"firstFramePath": "/abs/path/from/assets.json.png",
|
|
274
|
+
"lastFramePath": "/abs/optional.png",
|
|
275
|
+
"durationSec": 5
|
|
276
|
+
}]
|
|
277
|
+
}
|
|
278
|
+
```
|
|
279
|
+
|
|
280
|
+
Then call `video_render` with that path. Interrupted? Call it again with the
|
|
281
|
+
same path — it resumes. If an ambiguous submit is reported, do not delete a
|
|
282
|
+
shot or call render again blindly: run `/video-gen recover <jobId>`, check the
|
|
283
|
+
provider console, then explicitly `reset` a confirmed-absent task or `adopt`
|
|
284
|
+
its task id. Revisions? New job directory.
|