@amaster.ai/pi-video-gen 0.1.8 → 0.1.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +57 -111
- package/dist/config.d.ts.map +1 -1
- package/dist/config.js +9 -0
- package/dist/config.js.map +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +217 -40
- package/dist/index.js.map +1 -1
- package/dist/jobs/store.d.ts.map +1 -1
- package/dist/jobs/store.js +5 -7
- package/dist/jobs/store.js.map +1 -1
- package/dist/prompt.d.ts +28 -0
- package/dist/prompt.d.ts.map +1 -0
- package/dist/prompt.js +112 -0
- package/dist/prompt.js.map +1 -0
- package/dist/providers/ark.d.ts +2 -2
- package/dist/providers/ark.d.ts.map +1 -1
- package/dist/providers/ark.js +47 -4
- package/dist/providers/ark.js.map +1 -1
- package/dist/providers/dashscope.js +2 -2
- package/dist/providers/dashscope.js.map +1 -1
- package/dist/providers/kling.js +2 -2
- package/dist/providers/kling.js.map +1 -1
- package/dist/providers/minimax.js +2 -2
- package/dist/providers/minimax.js.map +1 -1
- package/dist/providers/models.d.ts.map +1 -1
- package/dist/providers/models.js +9 -0
- package/dist/providers/models.js.map +1 -1
- package/dist/providers/newapi.js +2 -2
- package/dist/providers/newapi.js.map +1 -1
- package/dist/providers/openrouter.js +2 -2
- package/dist/providers/openrouter.js.map +1 -1
- package/dist/providers/request.d.ts +11 -1
- package/dist/providers/request.d.ts.map +1 -1
- package/dist/providers/request.js +56 -0
- package/dist/providers/request.js.map +1 -1
- package/dist/providers/task.d.ts +4 -0
- package/dist/providers/task.d.ts.map +1 -1
- package/dist/providers/task.js +10 -2
- package/dist/providers/task.js.map +1 -1
- package/dist/render.d.ts +6 -4
- package/dist/render.d.ts.map +1 -1
- package/dist/render.js +53 -11
- package/dist/render.js.map +1 -1
- package/dist/types.d.ts +49 -0
- package/dist/types.d.ts.map +1 -1
- package/package.json +5 -5
- package/skills/video-gen/SKILL.md +90 -171
- package/skills/video-gen/evals.json +65 -0
- package/skills/video-gen/references/seedance-personas.json +72626 -0
- package/skills/video-gen/references/seedance-public-material-library.md +269 -0
- package/skills/video-gen/scripts/search-seedance-personas.mjs +75 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@amaster.ai/pi-video-gen",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.10",
|
|
4
4
|
"description": "Pi extension for AI video generation plus local video composition: lossless clip concat and mixed image/video timelines with overlays, TTS, soft or burned subtitles, source audio, BGM, and bundled LGPL/GPL FFmpeg runtimes.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package",
|
|
@@ -53,14 +53,14 @@
|
|
|
53
53
|
"dependencies": {
|
|
54
54
|
"sharp": "^0.34.0",
|
|
55
55
|
"msedge-tts": "^2.0.7",
|
|
56
|
-
"@amaster.ai/pi-shared": "0.1.
|
|
56
|
+
"@amaster.ai/pi-shared": "0.1.10"
|
|
57
57
|
},
|
|
58
58
|
"optionalDependencies": {
|
|
59
|
-
"@amaster.ai/pi-video-gen-ffmpeg-linux-arm64": "0.1.8",
|
|
60
59
|
"@amaster.ai/pi-video-gen-ffmpeg-darwin-arm64": "0.1.8",
|
|
61
|
-
"@amaster.ai/pi-video-gen-ffmpeg-darwin-x64": "0.1.8",
|
|
62
60
|
"@amaster.ai/pi-video-gen-ffmpeg-linux-x64": "0.1.8",
|
|
63
|
-
"@amaster.ai/pi-video-gen-ffmpeg-win32-x64": "0.1.8"
|
|
61
|
+
"@amaster.ai/pi-video-gen-ffmpeg-win32-x64": "0.1.8",
|
|
62
|
+
"@amaster.ai/pi-video-gen-ffmpeg-linux-arm64": "0.1.8",
|
|
63
|
+
"@amaster.ai/pi-video-gen-ffmpeg-darwin-x64": "0.1.8"
|
|
64
64
|
},
|
|
65
65
|
"peerDependencies": {
|
|
66
66
|
"@earendil-works/pi-ai": ">=0.80.10",
|
|
@@ -8,16 +8,11 @@ description: "Video creation and local composition: join existing clips or rende
|
|
|
8
8
|
This skill orchestrates four published flows:
|
|
9
9
|
|
|
10
10
|
- **C0 local concat** — `video_compose` joins compatible existing MP4 clips.
|
|
11
|
-
- **Timeline render** — `video_compose` locally turns images/screenshots and
|
|
12
|
-
existing video clips into a video with overlays, motion, transitions, soft or
|
|
13
|
-
burned subtitles, source audio, and optional BGM. Narration is an optional
|
|
14
|
-
network feature.
|
|
11
|
+
- **Timeline render** — `video_compose` locally turns images/screenshots and existing video clips into a video with overlays, motion, transitions, soft or burned subtitles, source audio, and optional BGM. Narration is an optional network feature.
|
|
15
12
|
- **Single AI clip** — one paid `video_generate` call.
|
|
16
|
-
- **Shot-book AI film** — author a shot book, generate frames with
|
|
17
|
-
`image_generate`, then ONE paid `video_render` call renders and stitches.
|
|
13
|
+
- **Shot-book AI film** — author a shot book, generate frames with `image_generate`, then ONE paid `video_render` call renders and stitches.
|
|
18
14
|
|
|
19
|
-
The AI video model is fixed by `pi-video-gen.defaultModel`; generated images use
|
|
20
|
-
pi-image-gen's active model.
|
|
15
|
+
The AI video model is fixed by `pi-video-gen.defaultModel`; generated images use pi-image-gen's active model.
|
|
21
16
|
|
|
22
17
|
## A. Workflow rules
|
|
23
18
|
|
|
@@ -30,82 +25,31 @@ pi-image-gen's active model.
|
|
|
30
25
|
| Multi-shot film with keyframes | `video_render` |
|
|
31
26
|
| Promo/explainer from images, screenshots & clips | `video_compose` (TimelineSpec — local mixed-media render; optional network TTS) |
|
|
32
27
|
|
|
33
|
-
1. **Local flows stop here.** For C0 follow §A0; for Timeline follow §A1. Do not
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
4. **Write the shot book in conversation** (schema in §B). If the user only has a
|
|
45
|
-
vague idea, first be the screenwriter: three-act structure, filmable actions
|
|
46
|
-
("show, don't tell"), concrete visual detail. Iterate with the user in chat.
|
|
47
|
-
5. **Confirmation gate 1 (shot-book only).** Show the shot-book summary — shot count,
|
|
48
|
-
character list, estimated image calls (~2N+3C) and video calls (N) — and get an
|
|
49
|
-
explicit go-ahead. **Default small: 1 scene, 3–5 shots** unless the user asks
|
|
50
|
-
for more.
|
|
51
|
-
6. **Image stage (shot-book only, via `image_generate`, per §C).** Character portraits →
|
|
52
|
-
per-shot first frame (and last frame when needed). Show each batch to the user.
|
|
53
|
-
7. **Paid confirmation.** Before `video_generate`, confirm its one paid call.
|
|
54
|
-
For a shot book, once frames are ready, state "about to make N paid
|
|
55
|
-
video calls" and get an explicit render order. Then assemble the render spec
|
|
56
|
-
and call `video_render` ONCE.
|
|
57
|
-
8. **Cost honesty.** AI video calls are paid and take minutes each. Never state
|
|
58
|
-
amounts (prices change); state call counts and durations.
|
|
59
|
-
9. **Revisions.** The render spec is immutable per job directory. Text-stage
|
|
60
|
-
revisions happen in chat (regenerate frames as needed); a revised film goes in
|
|
61
|
-
a NEW job directory. NEVER suggest "delete shots/<id>/ and rerender" — that
|
|
62
|
-
breaks downstream dependencies. Rerunning the SAME spec path resumes an
|
|
63
|
-
interrupted job (finished shots don't re-bill).
|
|
64
|
-
10. **Degradation negotiation.** If `video_render` preflight fails (e.g. last
|
|
65
|
-
frame unsupported), present the options (switch model / edit spec /
|
|
66
|
-
`allowDegradations`) and let the user choose. Never degrade silently. When the
|
|
67
|
-
model's `nativeAudio` is false, don't write audio cues into video prompts
|
|
68
|
-
unless the user accepted silence.
|
|
69
|
-
11. **Cancellation honesty.** Interrupting stops local polling only — remote
|
|
70
|
-
tasks may keep running and billable (Ark cancellation is unverified). Say so.
|
|
28
|
+
1. **Local flows stop here.** For C0 follow §A0; for Timeline follow §A1. Do not run the AI preflight, shot-book steps, or paid confirmation gates below. Timeline only needs `image_generate` when its source images do not already exist.
|
|
29
|
+
2. **AI preflight only.** For `video_generate` or `video_render`, call `video_capabilities` and respect the active model's duration range and audio support, frame support, and trusted asset modalities. Confirm `image_generate` is available only when source frames need to be generated (`/video-gen doctor` checks; config health is `/image-gen list`).
|
|
30
|
+
3. **Pick the AI flow.** A vague idea or a script that needs multiple shots → shot-book flow. One moving shot → `video_generate`. A still → pi-image-gen.
|
|
31
|
+
4. **Write the shot book in conversation** (schema in §B). If the user only has a vague idea, first be the screenwriter: three-act structure, filmable actions ("show, don't tell"), concrete visual detail. Iterate with the user in chat.
|
|
32
|
+
5. **Confirmation gate 1 (shot-book only).** Show the shot-book summary — shot count, character list, estimated image calls (~2N+3C) and video calls (N) — and get an explicit go-ahead. **Default small: 1 scene, 3–5 shots** unless the user asks for more.
|
|
33
|
+
6. **Source stage (shot-book only).** Use trusted portrait assets per §A2 when Seedance will receive a recognizable real person. Otherwise generate the required character portraits and per-shot frames via `image_generate` per §C. Show each generated batch to the user.
|
|
34
|
+
7. **Paid confirmation.** Before `video_generate`, confirm its one paid call. For a shot book, once frames are ready, state "about to make N paid video calls" and get an explicit render order. Then assemble the render spec and call `video_render` ONCE.
|
|
35
|
+
8. **Cost honesty.** AI video calls are paid and take minutes each. Never state amounts (prices change); state call counts and durations.
|
|
36
|
+
9. **Revisions.** The render spec is immutable per job directory. Text-stage revisions happen in chat (regenerate frames as needed); a revised film goes in a NEW job directory. NEVER suggest "delete shots/<id>/ and rerender" — that breaks downstream dependencies. Rerunning the SAME spec path resumes an interrupted job (finished shots don't re-bill).
|
|
37
|
+
10. **Degradation negotiation.** If `video_render` preflight fails (e.g. last frame unsupported), present the options (switch model / edit spec / `allowDegradations`) and let the user choose. Never degrade silently. When the model's `nativeAudio` is false, don't write audio cues into video prompts unless the user accepted silence.
|
|
38
|
+
11. **Cancellation honesty.** Interrupting stops local polling only — remote tasks may keep running and billable (Ark cancellation is unverified). Say so.
|
|
71
39
|
|
|
72
40
|
## A0. C0 — composing existing clips (`video_compose`)
|
|
73
41
|
|
|
74
|
-
1. **Tell the user first**: clip count, order, output location
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
under the video-gen output dir, then call `video_compose` ONCE. Keep source
|
|
80
|
-
clips outside `<jobDir>/clips/`; that directory and `final_video.mp4` are
|
|
81
|
-
reserved pipeline outputs, and a fresh job refuses either conflict.
|
|
82
|
-
3. **Only promise** lossless concat of compatible MP4s (C0). **Never promise**
|
|
83
|
-
trimming, transitions, overlays, subtitles, TTS, BGM, or re-encoding **for
|
|
84
|
-
the C0 path** — those live in the Timeline path (A1 below), not here; do
|
|
85
|
-
not hint at them for `compose-input.json`.
|
|
86
|
-
4. On any ordered stream incompatibility across all tracks
|
|
87
|
-
(codec/resolution/fps/timebase/pix_fmt/sample-rate/audio layout), hand the
|
|
88
|
-
exact ffprobe differences back to the user/agent:
|
|
89
|
-
re-encode the odd clips first. NEVER silently transcode, and NEVER fall
|
|
90
|
-
back to `video_generate`/`video_render` as a workaround.
|
|
91
|
-
5. Interrupted? Rerun the SAME path (fingerprint-verified resume / cached).
|
|
92
|
-
Changed clips or order? NEW job directory. A completed final video is
|
|
93
|
-
hash-bound; if it is missing or changed, restore the exact artifact or start
|
|
94
|
-
a NEW job.
|
|
42
|
+
1. **Tell the user first**: clip count, order, output location (`<jobDir>/final_video.mp4`), `mode: "copy"`. This is LOCAL compute — do not use the paid-model confirmation script for it.
|
|
43
|
+
2. Write `<jobDir>/compose-input.json` (`{"clips":[{"id":"c1","path":"/abs/a.mp4"},…],"output":{"mode":"copy"}}`) under the video-gen output dir, then call `video_compose` ONCE. Keep source clips outside `<jobDir>/clips/`; that directory and `final_video.mp4` are reserved pipeline outputs, and a fresh job refuses either conflict.
|
|
44
|
+
3. **Only promise** lossless concat of compatible MP4s (C0). **Never promise** trimming, transitions, overlays, subtitles, TTS, BGM, or re-encoding **for the C0 path** — those live in the Timeline path (A1 below), not here; do not hint at them for `compose-input.json`.
|
|
45
|
+
4. On any ordered stream incompatibility across all tracks (codec/resolution/fps/timebase/pix_fmt/sample-rate/audio layout), hand the exact ffprobe differences back to the user/agent: re-encode the odd clips first. NEVER silently transcode, and NEVER fall back to `video_generate`/`video_render` as a workaround.
|
|
46
|
+
5. Interrupted? Rerun the SAME path (fingerprint-verified resume / cached). Changed clips or order? NEW job directory. A completed final video is hash-bound; if it is missing or changed, restore the exact artifact or start a NEW job.
|
|
95
47
|
|
|
96
48
|
## A1. Timeline compose (`video_compose` with `timeline-input.json`)
|
|
97
49
|
|
|
98
|
-
Use for promos/explainers from still images and existing clips. Media rendering
|
|
99
|
-
|
|
100
|
-
the
|
|
101
|
-
using it.
|
|
102
|
-
|
|
103
|
-
1. **Collect existing images/screenshots/clips first**, and use `image_generate`
|
|
104
|
-
only for missing visual material. Keep all source media outside the job
|
|
105
|
-
directory, then author `<jobDir>/timeline-input.json`.
|
|
106
|
-
`assets/`, `overlays/`, `audio/`, `segments/`, `qc/`, generated tracks,
|
|
107
|
-
subtitles, and `final_video.mp4` are reserved pipeline outputs; a fresh job
|
|
108
|
-
refuses any conflicts rather than deleting them.
|
|
50
|
+
Use for promos/explainers from still images and existing clips. Media rendering is local and uses no paid video model. If narration is requested, disclose that the explicit `edge-tts:<voice>` option sends narration text to Microsoft before using it.
|
|
51
|
+
|
|
52
|
+
1. **Collect existing images/screenshots/clips first**, and use `image_generate` only for missing visual material. Keep all source media outside the job directory, then author `<jobDir>/timeline-input.json`. `assets/`, `overlays/`, `audio/`, `segments/`, `qc/`, generated tracks, subtitles, and `final_video.mp4` are reserved pipeline outputs; a fresh job refuses any conflicts rather than deleting them.
|
|
109
53
|
```jsonc
|
|
110
54
|
{
|
|
111
55
|
"title": "产品宣传片",
|
|
@@ -135,40 +79,29 @@ using it.
|
|
|
135
79
|
]
|
|
136
80
|
}
|
|
137
81
|
```
|
|
138
|
-
2. **Chinese text NEVER comes from an image model** — titles/subtitles go in
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
PNGs) before showing the result — flipped/overlapping text only shows up
|
|
162
|
-
visually. Soft `mov_text` subtitles are not burned into those PNGs; the
|
|
163
|
-
pipeline separately verifies that the subtitle stream exists and that the
|
|
164
|
-
SRT cues match the resolved segment timeline.
|
|
165
|
-
7. The spec is immutable per job: rerunning the same path resumes only
|
|
166
|
-
regular job-local artifacts whose manifest hashes still match; changes
|
|
167
|
-
require a NEW job directory. A committed artifact that is missing or
|
|
168
|
-
changed is rejected rather than regenerated underneath cached downstream
|
|
169
|
-
outputs. A completed manifest with any missing artifact hash is rejected;
|
|
170
|
-
an interrupted manifest invalidates unverified downstream hashes before
|
|
171
|
-
rebuilding an uncommitted upstream artifact.
|
|
82
|
+
2. **Chinese text NEVER comes from an image model** — titles/subtitles go in `overlay` and are rendered locally via SVG (no garbled CJK).
|
|
83
|
+
3. **Paid-model confirmation is unnecessary** for the local media render, but still show the segment count and total planned duration before calling `video_compose`. Obtain explicit agreement before sending narration text to Edge TTS.
|
|
84
|
+
4. Every segment contains exactly one of `image` or `video`. Video segments are normalized to the output resolution/fps, may be trimmed/scaled, and mix their source audio with narration before optional BGM. Video source audio without a stream degrades to silence; `sourceAudio.muted: true` or `volume: 0` disables it. A video's numeric `durationSec` is its fixed trim window; narration that does not fit is rejected instead of extending it.
|
|
85
|
+
5. Narration uses Edge TTS only with an explicit `edge-tts:<voice>` selection; its text is sent to Microsoft. Measured audio duration drives image `durationSec: "auto"`; subtitles use each segment's actual video timing. `subtitles.mode` defaults to `"soft"` (`mov_text`); `"burn"` renders the configured font/color/background directly into each narrated segment. TTS failures stop the job by default. Use `ttsFailureMode: "silent-subtitles"` only as an explicit degradation choice; it keeps the subtitle track and fills that segment with silence. Once accepted, that degradation is cached for the immutable job; create a NEW job to retry real narration.
|
|
86
|
+
6. On completion, review the QC frames in `<jobDir>/qc/` yourself (Read the PNGs) before showing the result — flipped/overlapping text only shows up visually. Soft `mov_text` subtitles are not burned into those PNGs; the pipeline separately verifies that the subtitle stream exists and that the SRT cues match the resolved segment timeline.
|
|
87
|
+
7. The spec is immutable per job: rerunning the same path resumes only regular job-local artifacts whose manifest hashes still match; changes require a NEW job directory. A committed artifact that is missing or changed is rejected rather than regenerated underneath cached downstream outputs. A completed manifest with any missing artifact hash is rejected; an interrupted manifest invalidates unverified downstream hashes before rebuilding an uncommitted upstream artifact.
|
|
88
|
+
|
|
89
|
+
## A2. Seedance trusted portraits and provider-managed assets
|
|
90
|
+
|
|
91
|
+
Before any paid Seedance call that may contain a recognizable human face:
|
|
92
|
+
|
|
93
|
+
1. **Classify the source.** Ordinary uploaded or generated images/videos of a recognizable real person may be rejected by Seedance's privacy checks. Do not treat a local file path or public URL as an authorized portrait.
|
|
94
|
+
2. **Search preset personas when needed.** If the user has not already chosen an identity, run this skill's `scripts/search-seedance-personas.mjs` with `--query "<space-separated traits>" --framing half|full --limit 5`. Use `half` for close/medium portrait shots and `full` when the whole body or body movement must be visible. Present the bounded matches with label, short bio, framing, and Asset ID; let the user choose before a paid call. The script is the access path; do not load the entire 3.6 MB JSON catalog into context.
|
|
95
|
+
3. **Use the exact trusted asset.** For a catalog choice, copy the returned `selectedAssetId`. Otherwise ask the user for a preset-avatar or Active authorized-person `asset-...` ID from the current Ark account/project. Catalog IDs were observed on 2026-08-24 and are not documented as permanent or cross-account. If the active account rejects one, ask the user to copy the current ID from that account's virtual-avatar library.
|
|
96
|
+
4. **Preserve modality order.** Put provider assets in `referenceAssets` as `{ "modality": "image|video|audio", "assetId": "asset-..." }`. Order is significant within each modality. In the prompt say `Image 1`, `Video 1`, or `Audio 1`; never expose or cite the Asset ID in prompt prose. For the built-in Seedance 2.0 models, keep each request within 9 image references total (local frames plus image assets), 3 video assets, and 3 audio assets.
|
|
97
|
+
5. **Reconfirm the paid context.** Approval is scoped to the exact asset list, provider/account context, model, clip count, and duration. A changed asset, account/project, provider, model, or job requires a new confirmation.
|
|
98
|
+
6. **Keep onboarding out of scope.** This package submits already-created assets. It does not perform identity verification, authorization H5 flows, asset activation, upload, or asset-library management.
|
|
99
|
+
|
|
100
|
+
Official references: [preset avatars](https://docs.volcengine.com/docs/82379/2608626?lang=zh#preset-avatar), [authorized-person assets](https://docs.volcengine.com/docs/82379/2223965?lang=zh), and [Seedance asset request format](https://console.volcengine.com/ark/region:cn-beijing/docs/82379/2333589?projectName=default&lang=zh#d9a7d853).
|
|
101
|
+
|
|
102
|
+
## A3. Seedance public image/video/audio materials
|
|
103
|
+
|
|
104
|
+
When the user asks for built-in Seedance materials, action references, camera references, visual styles, environments, characters, or sample voices, read [`references/seedance-public-material-library.md`](references/seedance-public-material-library.md) completely before proposing choices. Use the exact Chinese display labels from that catalog, copy its exact Asset IDs into `referenceAssets`, and keep the selected media in modality order. These IDs were read from the public material cards, not from the separate virtual-avatar library. If the active account rejects a listed ID, re-open the experience center and copy the current card ID instead of guessing or substituting a media URL.
|
|
172
105
|
|
|
173
106
|
## B. Shot book (VideoProject) — authoring reference
|
|
174
107
|
|
|
@@ -183,9 +116,12 @@ Author as JSON in conversation; save to `<jobDir>/project.json` for the record.
|
|
|
183
116
|
"shots": [{
|
|
184
117
|
"id": "s1",
|
|
185
118
|
"intent": "Wide shot, rainy alley. <Alice> enters from the left, stops under the streetlamp…",
|
|
119
|
+
"scene": "Rainy alley at night, neon signs, wet pavement", // optional: the shot's setting
|
|
186
120
|
"firstFrame": "…pure static description of the FIRST frame…",
|
|
187
121
|
"lastFrame": "…(optional) pure static description of the LAST frame…",
|
|
188
|
-
"
|
|
122
|
+
"visuals": "Static camera, wide shot from across the street…", // camera + framing only
|
|
123
|
+
"action": "A woman with long blonde hair and a red scarf walks in from the left…",
|
|
124
|
+
"effects": "…(optional) time-varying visuals: rain picks up, neon reflections intensify…",
|
|
189
125
|
"audio": "[Sound Effect] rain, distant traffic. [Speaker] Alice (soft): \"We're here.\"",
|
|
190
126
|
"visibleCharacters": ["alice"],
|
|
191
127
|
"durationSec": 5,
|
|
@@ -198,46 +134,25 @@ Author as JSON in conversation; save to `<jobDir>/project.json` for the record.
|
|
|
198
134
|
|
|
199
135
|
Field rules:
|
|
200
136
|
|
|
201
|
-
- **Every shot needs a narrative purpose** (establish / emotion / reaction). First
|
|
202
|
-
|
|
203
|
-
- **
|
|
204
|
-
|
|
205
|
-
- **
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
- **
|
|
209
|
-
|
|
210
|
-
- **lastFrame needed when**: composition/focus changes drastically, a character
|
|
211
|
-
enters or turns to face camera, a major reveal happens. Otherwise omit it.
|
|
212
|
-
- **Few camera positions.** Default: one `continuityGroup` for everything. New
|
|
213
|
-
group only when shot size/angle/focus differs significantly.
|
|
214
|
-
- **continuityGroup** = shots sharing a space/base image; **startFrameFromShotId**
|
|
215
|
-
pins a specific parent frame for composition; **continuityNote** says what the
|
|
216
|
-
parent frame lacks (the frame prompt must then keep the background and replace
|
|
217
|
-
those elements). Self-check: parent shot EXISTS, comes EARLIER, same
|
|
218
|
-
continuityGroup, no cycles.
|
|
137
|
+
- **Every shot needs a narrative purpose** (establish / emotion / reaction). First shot: widest view of the scene. Close-ups for emotion, wide shots for context.
|
|
138
|
+
- **At most one dialogue line per shot.** Character names in `intent` are wrapped in angle brackets: `<Alice>`.
|
|
139
|
+
- **firstFrame / lastFrame are pure static snapshots** — no ongoing actions ("he is sitting, leaning forward", NOT "he is about to stand"). Include shot size, angle, composition, who is where and facing which way.
|
|
140
|
+
- **visuals = camera + framing only** (movement, shot size, angle, focus); **action = in-frame movement only**. Split them — they become separate labeled sections in the assembled prompt. Refer to characters by visible traits ("the woman in the red scarf"), never by name.
|
|
141
|
+
- **scene is the shot's setting**, copied verbatim into the render spec's `prompt.scene`. Optional when the first frame fully anchors the setting; write it when the setting carries mood/lighting the frame may not convey.
|
|
142
|
+
- **effects is for what a static frame cannot carry**: transformations, lighting/atmosphere shifts, particles, slow motion. Omit when the shot is visually static.
|
|
143
|
+
- **lastFrame needed when**: composition/focus changes drastically, a character enters or turns to face camera, a major reveal happens. Otherwise omit it.
|
|
144
|
+
- **Few camera positions.** Default: one `continuityGroup` for everything. New group only when shot size/angle/focus differs significantly.
|
|
145
|
+
- **continuityGroup** = shots sharing a space/base image; **startFrameFromShotId** pins a specific parent frame for composition; **continuityNote** says what the parent frame lacks (the frame prompt must then keep the background and replace those elements). Self-check: parent shot EXISTS, comes EARLIER, same continuityGroup, no cycles.
|
|
219
146
|
- **audio** uses `[Sound Effect] …` / `[Speaker] Name (Emotion): "line"` format.
|
|
220
|
-
- **durationSec and all capability values come from `video_capabilities`** —
|
|
221
|
-
|
|
222
|
-
frame support differ per model and change over time.
|
|
223
|
-
- **Behavioral quirks worth knowing** (still verify with `video_capabilities`):
|
|
224
|
-
some models have no native audio (omit audio cues or the render is silent);
|
|
225
|
-
some cannot do last-frame interpolation (never pass lastFrame to them);
|
|
226
|
-
HappyHorse takes a first frame OR reference images in one call, not both —
|
|
227
|
-
cite references in the prompt as `[Image 1]`, `[Image 2]`, …
|
|
147
|
+
- **durationSec and all capability values come from `video_capabilities`** — never from memory or this document. Durations, resolutions, ratios, audio and frame support differ per model and change over time.
|
|
148
|
+
- **Behavioral quirks worth knowing** (still verify with `video_capabilities`): some models have no native audio (omit audio cues or the render is silent); some cannot do last-frame interpolation (never pass lastFrame to them); HappyHorse takes a first frame OR reference images in one call, not both — cite references in the prompt as `[Image 1]`, `[Image 2]`, …
|
|
228
149
|
|
|
229
150
|
## C. Image operation manual (via `image_generate`)
|
|
230
151
|
|
|
231
|
-
Generic `image_generate` usage (params, sizes, `n`, edit labeling) follows the
|
|
232
|
-
**pi-image-gen skill** — it is the single authority; do not deviate. Two
|
|
233
|
-
video-specific handoff rules:
|
|
152
|
+
Generic `image_generate` usage (params, sizes, `n`, edit labeling) follows the **pi-image-gen skill** — it is the single authority; do not deviate. Two video-specific handoff rules:
|
|
234
153
|
|
|
235
|
-
- **Never assume a saved filename**: the actual extension follows the MIME type
|
|
236
|
-
|
|
237
|
-
record it immediately in `assets.json` (see below) and reference it in the
|
|
238
|
-
render spec.
|
|
239
|
-
- `assets.json` in the job dir: `{ "assets": { "<shotId>/<part>": { "sourcePath": "…" } } }`
|
|
240
|
-
mapping semantic assets (e.g. `s1/firstFrame`, `alice/front`) to real paths.
|
|
154
|
+
- **Never assume a saved filename**: the actual extension follows the MIME type and collisions get `-v2`. **The returned absolute path is the only truth** — record it immediately in `assets.json` (see below) and reference it in the render spec.
|
|
155
|
+
- `assets.json` in the job dir: `{ "assets": { "<shotId>/<part>": { "sourcePath": "…" } } }` mapping semantic assets (e.g. `s1/firstFrame`, `alice/front`) to real paths.
|
|
241
156
|
|
|
242
157
|
**Character portraits (3 views per visible character)**:
|
|
243
158
|
|
|
@@ -245,25 +160,16 @@ video-specific handoff rules:
|
|
|
245
160
|
- side (edit with front as reference): `Generate a full-body, side-view portrait of character {identifier} based on the provided front-view portrait, with a pure white background. Use a wide 16:9 landscape canvas, not a vertical portrait canvas. The character should be centered in the image, occupying the middle of the wide frame with enough horizontal empty space. Facing left. Standing with arms relaxed at sides.`
|
|
246
161
|
- back (edit with front as reference): `Generate a full-body, back-view portrait of character {identifier} based on the provided front-view portrait, with a pure white background. Use a wide 16:9 landscape canvas, not a vertical portrait canvas. The character should be centered in the image, occupying the middle of the wide frame with enough horizontal empty space. No facial features should be visible.`
|
|
247
162
|
|
|
248
|
-
If side/back fails after one retry, reuse front. Characters with `visible: false`
|
|
249
|
-
get no portraits.
|
|
163
|
+
If side/back fails after one retry, reuse front. Characters with `visible: false` get no portraits.
|
|
250
164
|
|
|
251
|
-
**Reference selection for frames**:
|
|
252
|
-
candidates = portraits of visible characters (ONE view each, chosen by facing) +
|
|
253
|
-
continuity frames. Pick a SMALL set of the most relevant ones — same
|
|
254
|
-
camera/group first, most recent frames first, drop redundant near-duplicates,
|
|
255
|
-
prefer the portrait when a character newly appears. How many images a call
|
|
256
|
-
accepts is pi-image-gen's authority (its skill/tool description), not this
|
|
257
|
-
document's.
|
|
165
|
+
**Reference selection for frames**: candidates = portraits of visible characters (ONE view each, chosen by facing) + continuity frames. Pick a SMALL set of the most relevant ones — same camera/group first, most recent frames first, drop redundant near-duplicates, prefer the portrait when a character newly appears. How many images a call accepts is pi-image-gen's authority (its skill/tool description), not this document's.
|
|
258
166
|
|
|
259
|
-
**Frame prompt assembly**: prefix each reference image with its role, then the
|
|
260
|
-
frame description mapping elements to images:
|
|
167
|
+
**Frame prompt assembly**: prefix each reference image with its role, then the frame description mapping elements to images:
|
|
261
168
|
|
|
262
169
|
```
|
|
263
170
|
Image 0: A front view portrait of Alice.
|
|
264
171
|
Image 1: [alley] Wide shot of the rainy alley from shot s1.
|
|
265
|
-
Create an image based on the following description: <firstFrame text>. The alley
|
|
266
|
-
background should reference Image 1; Alice's appearance should reference Image 0.
|
|
172
|
+
Create an image based on the following description: <firstFrame text>. The alley background should reference Image 1; Alice's appearance should reference Image 0.
|
|
267
173
|
```
|
|
268
174
|
|
|
269
175
|
## D. Assemble the render spec and render
|
|
@@ -273,22 +179,35 @@ Write `<outputDir>/<jobId>/render-input.json` (jobId: letters/digits/dash/unders
|
|
|
273
179
|
```jsonc
|
|
274
180
|
{
|
|
275
181
|
"title": "…", "aspectRatio": "16:9",
|
|
182
|
+
"style": "Cartoon, warm palette, soft shading", // film-level look — from the shot book's style
|
|
183
|
+
"characters": [ // film-level registry — id + appearance/outfit merged
|
|
184
|
+
{ "id": "alice", "description": "long blonde hair, blue eyes, slender; red scarf, black leather jacket" }
|
|
185
|
+
],
|
|
186
|
+
"consistency": "Faces, hair and outfits stay identical across the shot, no morphing or drift.",
|
|
187
|
+
"negative": "no text, watermarks, or subtitles",
|
|
276
188
|
"shots": [{
|
|
277
189
|
"id": "s1",
|
|
278
|
-
"
|
|
279
|
-
|
|
190
|
+
"prompt": { // from the shot book's fields, verbatim
|
|
191
|
+
"scene": "Rainy alley at night, neon signs, wet pavement", // omit when the first frame says it all
|
|
192
|
+
"visuals": "Static camera, wide shot from across the street",
|
|
193
|
+
"action": "The woman in the red scarf walks in from the left, stops under the streetlamp",
|
|
194
|
+
"effects": "Rain picks up halfway through; neon reflections ripple in puddles",
|
|
195
|
+
"audio": "[Sound Effect] rain, distant traffic. [Speaker] Alice (soft): \"We're here.\"",
|
|
196
|
+
"visibleCharacters": ["alice"]
|
|
197
|
+
},
|
|
198
|
+
"firstFramePath": "<project>/path/from/assets.json.png", // optional for asset-only shots
|
|
280
199
|
"lastFramePath": "<project>/optional.png",
|
|
200
|
+
"referenceAssets": [
|
|
201
|
+
{ "modality": "image", "assetId": "asset-avatar-from-current-account" },
|
|
202
|
+
{ "modality": "audio", "assetId": "asset-voice-from-current-account" }
|
|
203
|
+
],
|
|
281
204
|
"durationSec": 5
|
|
282
205
|
}]
|
|
283
206
|
}
|
|
284
207
|
```
|
|
285
208
|
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
approved project directory; symlinks and outside paths are rejected.
|
|
209
|
+
The plugin assembles each shot's labeled prompt (`[Style]` / `[Character]` / `[Scene]` / `[Visuals]` / `[Action]` / `[Effects]` / `[Audio]` + consistency and negative directives) — never pre-join a prompt string yourself. Validation fails the run before any paid call when: `visuals`/`action` are empty, `visibleCharacters` references an id missing from `characters`, or a shot without `firstFramePath` lacks `style`/`scene` (a frameless request must not go out action-only, including an asset-only request). Film-level `style`/`consistency`/`negative` apply to every shot — write them once; shot-level `scene` repeats per shot even when consecutive shots share a location (each shot is submitted independently).
|
|
210
|
+
|
|
211
|
+
Every reference-frame path that is present must resolve to a regular png/jpg/webp file inside the session cwd. Absolute paths are accepted only when they remain inside that approved project directory; symlinks and outside paths are rejected. `referenceAssets` are provider-managed and are not local files. Keep their order stable: changing an asset, modality, or order changes the request and requires a new job directory and paid confirmation.
|
|
289
212
|
|
|
290
|
-
Then call `video_render` with that path. Interrupted? Call it again with the
|
|
291
|
-
same path — it resumes. If an ambiguous submit is reported, do not delete a
|
|
292
|
-
shot or call render again blindly: run `/video-gen recover <jobId>`, check the
|
|
293
|
-
provider console, then explicitly `reset` a confirmed-absent task or `adopt`
|
|
294
|
-
its task id. Revisions? New job directory.
|
|
213
|
+
Then call `video_render` with that path. Interrupted? Call it again with the same path — it resumes. If an ambiguous submit is reported, do not delete a shot or call render again blindly: run `/video-gen recover <jobId>`, check the provider console, then explicitly `reset` a confirmed-absent task or `adopt` its task id. Revisions? New job directory.
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
{
|
|
2
|
+
"skill_name": "video-gen",
|
|
3
|
+
"evals": [
|
|
4
|
+
{
|
|
5
|
+
"id": "compatible-mp4-concat",
|
|
6
|
+
"prompt": "I have three compatible MP4 clips and only want them joined in order without transitions, overlays, trimming, or re-encoding. Describe the exact Pi video workflow, output mode, and whether a paid-model confirmation is needed. Do not call tools or create files.",
|
|
7
|
+
"expected_output": "Route to the local video_compose C0 flow with copy mode and no paid-model confirmation.",
|
|
8
|
+
"expectations": [
|
|
9
|
+
{
|
|
10
|
+
"text": "Routes to video_compose",
|
|
11
|
+
"includes": ["video_compose"]
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"text": "Uses lossless copy mode",
|
|
15
|
+
"includes_any": ["mode: copy", "mode `copy`", "\"copy\""]
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
"text": "Treats the operation as local rather than paid generation",
|
|
19
|
+
"includes": ["local"],
|
|
20
|
+
"includes_any": ["no paid", "does not need paid", "no paid-model", "not needed"]
|
|
21
|
+
}
|
|
22
|
+
]
|
|
23
|
+
},
|
|
24
|
+
{
|
|
25
|
+
"id": "single-ai-shot",
|
|
26
|
+
"prompt": "Create one five-second AI-generated shot of a paper boat moving through a rainy gutter. No multi-shot film is needed. Explain which Pi video flow and preflight you would use and what confirmation is required, but do not call tools.",
|
|
27
|
+
"expected_output": "Use video_capabilities, then one video_generate call after explicit paid confirmation.",
|
|
28
|
+
"expectations": [
|
|
29
|
+
{
|
|
30
|
+
"text": "Chooses the single-shot generation tool",
|
|
31
|
+
"includes": ["video_generate"]
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"text": "Runs capability preflight",
|
|
35
|
+
"includes": ["video_capabilities"]
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"text": "Requires confirmation before the paid call",
|
|
39
|
+
"includes": ["paid"],
|
|
40
|
+
"includes_any": ["confirmation", "confirm", "go-ahead", "approval"]
|
|
41
|
+
}
|
|
42
|
+
]
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"id": "multi-shot-film",
|
|
46
|
+
"prompt": "I want a short three-shot AI film with a recurring character and visual continuity between shots. Describe the planning, frame, confirmation, and render stages in Pi, including how many times the final render tool is invoked. Do not call tools.",
|
|
47
|
+
"expected_output": "Use a shot book, generate character/shot frames with image_generate, obtain confirmations, then call video_render once.",
|
|
48
|
+
"expectations": [
|
|
49
|
+
{
|
|
50
|
+
"text": "Uses a shot book for the multi-shot plan",
|
|
51
|
+
"includes_any": ["shot book", "shot-book"]
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"text": "Generates frames through image_generate",
|
|
55
|
+
"includes": ["image_generate"]
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"text": "Calls video_render only once",
|
|
59
|
+
"includes": ["video_render"],
|
|
60
|
+
"includes_any": ["call once", "once", "one call"]
|
|
61
|
+
}
|
|
62
|
+
]
|
|
63
|
+
}
|
|
64
|
+
]
|
|
65
|
+
}
|