@plaintake/scenario 1.24.0 → 1.30.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -58,6 +58,16 @@ The PlainTake CLI discovers and records files matching `*.demo.ts`. See the
58
58
  assertions, masks, waits, and handoffs to a person for steps PlainTake cannot perform itself
59
59
  (entering a one-time code, for example).
60
60
 
61
+ A scenario can also declare `terminal` to record a live command-line program (`run` is then
62
+ handed a `term` with `run`, `type`, `press`, `waitForText` and friends), and with
63
+ `terminal.browser: true` it can take turns between the terminal and a web app:
64
+ `demo.turn(term.actor)` and `demo.turn(web)` switch with a hard cut by default, and
65
+ `demo.turn(actor, { card })` or `{ card: false }` decides per turn. See the scenario guide.
66
+
67
+ For TikTok and YouTube Shorts, a scenario can declare a `hook` — `'3 PDFs → 1, free'` or
68
+ `{ text, durationMs }` — drawn above the picture for the video's first seconds when it is
69
+ recorded with `--for tiktok` or `--for shorts`. One or two lines; emoji are drawn as outlines.
70
+
61
71
  ## Licence
62
72
 
63
73
  This package (`@plaintake/scenario`) is MIT licensed — see [LICENSE](./LICENSE).
package/dist/index.js CHANGED
@@ -114,7 +114,7 @@ var DemoEventTypeSchema = z2.enum([
114
114
  "turn",
115
115
  /*
116
116
  * A full-frame motion-graphics cut-away: `demo.explain(...)` stops the screencast, splices
117
- * in a PlainMotion-rendered scene at this point on the timeline, and resumes the recording
117
+ * in a rendered motion-graphics scene at this point on the timeline, and resumes the recording
118
118
  * on the same page, same actor. A member of its own for the same reason `turn` is one: a
119
119
  * cut-away is not a step — no target, no verb, no hold — and its `payload` carries the
120
120
  * whole authored request (id, narration, scene props, cues) because the scene is compiled
@@ -161,6 +161,13 @@ var ScenarioIntroSchema = z3.object({
161
161
  narration: z3.string().min(1).optional(),
162
162
  durationMs: z3.number().int().positive().max(15e3, "intro.durationMs must be at most 15000ms").optional()
163
163
  });
164
+ var ScenarioHookSchema = z3.union([
165
+ z3.string().min(1),
166
+ z3.object({
167
+ text: z3.string().min(1),
168
+ durationMs: z3.number().int().min(1e3, "hook.durationMs must be at least 1000ms").max(1e4, "hook.durationMs must be at most 10000ms \u2014 past that it is a title, not a hook").optional()
169
+ })
170
+ ]);
164
171
  var ScenarioCameraSchema = z3.object({
165
172
  maxZoom: z3.number().min(1, "camera.maxZoom must be at least 1 (no zoom)").max(1.6, "camera.maxZoom must be at most 1.6 \u2014 beyond that the upscaled capture turns UI text to mush").optional(),
166
173
  margin: z3.number().min(0, "camera.margin must be at least 0").max(2, "camera.margin must be at most 2 (as a fraction of the target, per side)").optional(),
@@ -197,6 +204,30 @@ var ScenarioSpeechSchema = z3.object({
197
204
  */
198
205
  voices: z3.array(z3.string().min(1, "speech.voices entries must be non-empty voice names")).min(1, "speech.voices must name at least one voice \u2014 an empty list declares nothing, so omit it").optional()
199
206
  });
207
+ var EnvName = z3.string().regex(/^[A-Za-z_][A-Za-z0-9_]*$/);
208
+ var TerminalMetaSchema = z3.object({
209
+ shell: z3.enum(["bash", "zsh", "sh"]).default("bash"),
210
+ command: z3.array(z3.string().min(1)).min(1).optional(),
211
+ cols: z3.number().int().min(40).max(200).default(120),
212
+ rows: z3.number().int().min(12).max(60).default(34),
213
+ cwd: z3.string().min(1).optional(),
214
+ env: z3.array(EnvName).default([]),
215
+ secrets: z3.array(EnvName).default([]),
216
+ browser: z3.boolean().optional(),
217
+ label: z3.string().trim().min(1).optional(),
218
+ transition: z3.enum(["cut", "card"]).optional()
219
+ }).superRefine((t, ctx) => {
220
+ for (const field of ["label", "transition"]) {
221
+ if (t[field] !== void 0 && t.browser !== true) {
222
+ ctx.addIssue({ code: "custom", path: [field], message: `terminal.${field} needs terminal.browser: true` });
223
+ }
224
+ }
225
+ for (const name of t.secrets) {
226
+ if (!t.env.includes(name)) {
227
+ ctx.addIssue({ code: "custom", path: ["secrets"], message: `secret ${name} must also be listed in terminal.env` });
228
+ }
229
+ }
230
+ });
200
231
  var ScenarioMetaSchema = z3.object({
201
232
  schema: z3.literal("agent-demo.scenario/v1"),
202
233
  id: z3.string().regex(/^[a-z0-9][a-z0-9-]*$/, "must be a lowercase kebab-case id"),
@@ -207,6 +238,21 @@ var ScenarioMetaSchema = z3.object({
207
238
  height: z3.literal(1080),
208
239
  deviceScaleFactor: z3.literal(1)
209
240
  }),
241
+ /*
242
+ * How large the page is drawn inside the fixed 1920x1080 capture. At `s`, the page lays out
243
+ * at a 1920/s x 1080/s CSS viewport with deviceScaleFactor `s`, so Chromium paints every
244
+ * CSS pixel as s x s device pixels: the frame is still 1920x1080, the UI in it is s times
245
+ * larger and sharp, and the app's own responsive layout sees the narrower viewport. It is
246
+ * what a person gets from browser zoom, and what a vertical (`--aspect 9:16`) cut needs to
247
+ * stay legible once the whole frame is letterboxed to 1080 wide.
248
+ *
249
+ * `viewport` above still describes the capture, not the layout, which is why it stays a
250
+ * literal. Only 1.5 and 2 are offered besides 1: both divide 1920x1080 into whole CSS
251
+ * pixels, and both were measured (`ui-scale-dpr-spike.spec.ts`) to capture device-pixel
252
+ * frames headless and headed. Optional and inert when absent: every scenario written
253
+ * before this records exactly as it always did.
254
+ */
255
+ uiScale: z3.union([z3.literal(1), z3.literal(1.5), z3.literal(2)]).optional(),
210
256
  locale: z3.literal("en-US"),
211
257
  timezoneId: z3.literal("UTC"),
212
258
  colorScheme: z3.literal("light"),
@@ -253,6 +299,12 @@ var ScenarioMetaSchema = z3.object({
253
299
  * discarded in silence rather than refused.
254
300
  */
255
301
  intro: ScenarioIntroSchema.optional(),
302
+ /*
303
+ * The hook, drawn over the top band of a portrait render. Optional on the same precedent as
304
+ * `intro`; a scenario without one renders exactly what it always did, and a 16:9 render of one
305
+ * that has it carries it undrawn.
306
+ */
307
+ hook: ScenarioHookSchema.optional(),
256
308
  /*
257
309
  * Framing overrides for this scenario's pages. Optional like everything above it, and
258
310
  * inert when absent: the defaults are the constants the camera has always used.
@@ -268,6 +320,8 @@ var ScenarioMetaSchema = z3.object({
268
320
  * comment for the field's shape and the reasoning behind `speed`'s bounds.
269
321
  */
270
322
  speech: ScenarioSpeechSchema.optional(),
323
+ /* A live terminal as the recorded source instead of a web page; see `TerminalMetaSchema`. */
324
+ terminal: TerminalMetaSchema.optional(),
271
325
  /*
272
326
  * A text-substitution dictionary applied to narration immediately before synthesis — not the
273
327
  * vendored en-us phoneme dictionary `install-voice` fetches (a different thing entirely: that
@@ -307,10 +361,10 @@ var BundleManifestSchema = z5.object({
307
361
  fontSha256: Sha2562,
308
362
  containerImage: z5.string().optional(),
309
363
  /**
310
- * The plainmotion CLI's own `--version` string, present exactly when a scenario called
311
- * `demo.explain()` at least once — a run that never cut away never spawned it, and
312
- * `containerImage`'s own optionality is the precedent for saying so with an absent field
313
- * rather than a fabricated one.
364
+ * The plainmotion CLI's own `--version` string, present on bundles recorded by 1.28.0 or earlier
365
+ * whose scenario called `demo.explain()` — explain scenes were then rendered by that
366
+ * external CLI. Read-only now: explain scenes render in process, so nothing writes it, and
367
+ * it stays optional so those bundles keep verifying.
314
368
  */
315
369
  plainmotion: z5.string().optional()
316
370
  }),
@@ -320,8 +374,19 @@ var BundleManifestSchema = z5.object({
320
374
  architecture: z5.string(),
321
375
  locale: z5.literal("en-US"),
322
376
  timezone: z5.literal("UTC"),
377
+ /*
378
+ * The captured frame: 1920x1080 pixels at one pixel per video pixel, always. `uiScale`
379
+ * below says how the page was laid out into that frame; it never changes these two.
380
+ */
323
381
  viewport: z5.tuple([z5.literal(1920), z5.literal(1080)]),
324
- deviceScaleFactor: z5.literal(1)
382
+ deviceScaleFactor: z5.literal(1),
383
+ /*
384
+ * The scenario's `uiScale`, written only when it is not 1 — so every manifest recorded
385
+ * before the option existed, and every one recorded without it, stays byte-identical.
386
+ * Recorded because it changes what every target rect in the bundle was measured against
387
+ * (CSS pixels times this), which a diff between two bundles must not mistake for drift.
388
+ */
389
+ uiScale: z5.union([z5.literal(1.5), z5.literal(2)]).optional()
325
390
  }),
326
391
  assertions: z5.array(z5.object({ id: z5.string().min(1), status: z5.enum(["passed", "failed"]) })),
327
392
  artifacts: z5.array(
@@ -376,8 +441,26 @@ var CueSchema = z7.object({
376
441
  id: z7.string().min(1),
377
442
  startMs: z7.number().int().nonnegative(),
378
443
  endMs: z7.number().int().positive(),
379
- lines: z7.array(z7.string()).min(1).max(2)
380
- });
444
+ lines: z7.array(z7.string()).min(1).max(2),
445
+ /**
446
+ * When each word of `lines` is spoken, in video milliseconds — one entry per space-separated
447
+ * word of the lines joined, in order. Frozen only by a narrated recording under the social
448
+ * layout, whose captions light each word as the voice reaches it (`toAss`); absent everywhere
449
+ * else, so every other plan's cues serialise exactly as they did.
450
+ *
451
+ * Times, never invented: they are the synthesiser's own word timestamps, mapped back through
452
+ * the pronunciation dictionary onto the caption's words. A cue whose words those timestamps
453
+ * do not describe gets no `words` at all and is drawn whole, which is what a silent recording
454
+ * gets too.
455
+ */
456
+ words: z7.array(z7.number().int().nonnegative()).min(1).optional()
457
+ }).refine(
458
+ ({ words, lines, startMs, endMs }) => words === void 0 || words.length === lines.join(" ").split(" ").filter((word) => word !== "").length && words.every((at, index) => at >= startMs && at < endMs && (index === 0 || at >= words[index - 1])),
459
+ {
460
+ message: "words must time every word of the cue, in order, inside the cue",
461
+ path: ["words"]
462
+ }
463
+ );
381
464
  var HEX_COLOUR = z7.string().regex(/^#[0-9A-Fa-f]{6}$/, "must be a #RRGGBB colour");
382
465
  var OutroSchema = z7.object({
383
466
  durationMs: z7.number().int().positive().max(15e3),
@@ -393,6 +476,24 @@ var IntroSchema = z7.object({
393
476
  textColor: HEX_COLOUR,
394
477
  assPath: z7.literal("captions/intro.ass")
395
478
  });
479
+ var HookSchema = z7.object({
480
+ lines: z7.array(z7.string().min(1)).min(1).max(2),
481
+ durationMs: z7.number().int().min(1e3).max(1e4),
482
+ assPath: z7.literal("captions/hook.ass"),
483
+ /**
484
+ * The hook drawn as a picture rather than as ASS: a transparent PNG the width of the output
485
+ * and as tall as the bottom of its band, rasterised in Chromium when the hook was set (at
486
+ * record time, or by a `render --hook`/`--aspect` re-cut) so its emoji are drawn in colour by
487
+ * the system's emoji face — libass can only draw them as outlines. Frozen as a source file:
488
+ * a plain re-render overlays it and never opens a browser. Absent, the hook is drawn from
489
+ * `assPath` exactly as it was before 1.30, so older bundles re-render byte-identically.
490
+ */
491
+ imagePath: z7.literal("captions/hook.png").optional()
492
+ });
493
+ var FrameSchema = z7.object({
494
+ radius: z7.number().int().positive().max(96),
495
+ assPath: z7.literal("captions/frame.ass")
496
+ });
396
497
  var ChapterMarkSchema = z7.object({
397
498
  startMs: z7.number().int().nonnegative(),
398
499
  endMs: z7.number().int().positive(),
@@ -451,7 +552,17 @@ var CursorPointSchema = z7.object({
451
552
  * always did — so a cursor bundle from before the camera, or from before the letterbox,
452
553
  * still parses and regenerates byte-for-byte.
453
554
  */
454
- scale: z7.number().int().min(1).max(400).optional()
555
+ scale: z7.number().int().min(1).max(400).optional(),
556
+ /**
557
+ * Present only on the first point after a hard cut (`TransitionCutSchema`): the output
558
+ * instant the picture changes to the next actor's segment. The pointer then does not glide
559
+ * from the previous point — it parks there until `cutMs` and is parked on this point from
560
+ * `cutMs` on, so it never slides from a terminal row into a web page over live frames, nor
561
+ * moves over a frozen picture. Bounded by the previous point's `departMs` and this point's
562
+ * `arriveMs` (`CursorSchema`'s refine). Absent everywhere else, so a cut-free plan's points
563
+ * are byte-identical.
564
+ */
565
+ cutMs: z7.number().int().nonnegative().optional()
455
566
  });
456
567
  var CursorSchema = z7.object({
457
568
  assPath: z7.literal("captions/cursor.ass"),
@@ -467,6 +578,16 @@ var CursorSchema = z7.object({
467
578
  message: "points must be ordered, arrive before departing, and not overlap",
468
579
  path: ["points"]
469
580
  }
581
+ ).refine(
582
+ ({ points }) => points.every(
583
+ (point, index) => point.cutMs === void 0 || index > 0 && points[index - 1].departMs <= point.cutMs && point.cutMs <= point.arriveMs
584
+ ),
585
+ {
586
+ // A jump needs somewhere to jump from, and it must happen between leaving the previous
587
+ // point and reaching this one — anything else would show two arrows or none.
588
+ message: "cutMs must fall within [previous point departMs, arriveMs], never on the first point",
589
+ path: ["points"]
590
+ }
470
591
  );
471
592
  var CameraShotSchema = z7.object({
472
593
  id: z7.string().min(1),
@@ -515,7 +636,22 @@ var SegmentSchema = z7.object({
515
636
  id: z7.string().min(1),
516
637
  actorId: z7.string().min(1),
517
638
  startMs: z7.number().int().nonnegative(),
518
- endMs: z7.number().int().positive()
639
+ endMs: z7.number().int().positive(),
640
+ /**
641
+ * Where this segment's frames actually sit inside the concatenated `raw/session.webm`,
642
+ * measured from the segment files themselves (`probeSegmentSpanMs`, last packet's pts plus
643
+ * its duration — the offset the concat demuxer gives the next file). Present only on plans
644
+ * with a hard cut or a `terminal.browser` scenario, and then on every segment (`RenderPlanSchema`'s refine): the splice trims
645
+ * `[sourceStartMs, sourceEndMs)` instead of the capture-derived `[startMs, endMs)`, because
646
+ * Chromium's WebM writer can end a segment file short of its capture window, or round a
647
+ * sub-second capture up to a 1 s file, and a cut's freeze would otherwise clone a frame of
648
+ * the wrong actor. `sourceEndMs` is the usable end — the file's span capped at `captureMs` for
649
+ * every segment but the last (which reads past its file's end into the head tail-pad clone, as
650
+ * the capture-based trim always has) — so it may stop short of the next `sourceStartMs`.
651
+ * `startMs`/`endMs` keep their capture-derived values regardless.
652
+ */
653
+ sourceStartMs: z7.number().int().nonnegative().optional(),
654
+ sourceEndMs: z7.number().int().positive().optional()
519
655
  });
520
656
  var TransitionCardSchema = z7.object({
521
657
  fromActorId: z7.string().min(1),
@@ -524,12 +660,45 @@ var TransitionCardSchema = z7.object({
524
660
  lines: z7.array(z7.string().min(1)).min(1).max(2),
525
661
  backgroundColor: HEX_COLOUR,
526
662
  textColor: HEX_COLOUR,
527
- assPath: z7.string().regex(/^captions\/turn-[0-9]+\.ass$/)
663
+ assPath: z7.string().regex(/^captions\/turn-[0-9]+\.ass$/),
664
+ /**
665
+ * The outgoing segment's last-frame hold before this card, exactly as on a cut
666
+ * (`TransitionCutSchema.freezeMs`): `contentMs − trimmedSpanMs(source window)`. Only plans
667
+ * with measured source bounds (every plan with a cut, and every `terminal.browser` plan) carry it; a card in any other plan
668
+ * never does, so it parses and serialises as it always has.
669
+ */
670
+ freezeMs: z7.number().int().positive().optional()
528
671
  });
672
+ var TransitionCutSchema = z7.object({
673
+ kind: z7.literal("cut"),
674
+ fromActorId: z7.string().min(1),
675
+ toActorId: z7.string().min(1),
676
+ /**
677
+ * How long the outgoing segment's last frame is held before the cut (`tpad=stop_mode=clone`),
678
+ * so the video reaches the joint the derivation placed at the segment's content length:
679
+ * `contentMs − trimmedSpanMs(source window)`, present only when that is at least one frame —
680
+ * usually a trailing cue's overrun, but possibly a single frame (33ms) with no overrun at
681
+ * all, when the trim keeps one frame fewer than the content slot reserves. The joint itself
682
+ * still claims 0 (`transitionGapMs`). Cards and explain scenes in a source-bound plan (a cut, or `terminal.browser`) carry the
683
+ * same optional `freezeMs`; in a plan without source bounds nothing carries one.
684
+ */
685
+ freezeMs: z7.number().int().positive().optional()
686
+ }).refine((cut) => cut.fromActorId !== cut.toActorId, {
687
+ // The same reason `runtime.ts` refuses a turn to the actor already holding the browser:
688
+ // a joint where nobody changes is an explain scene's, never a hand-off's.
689
+ message: "a cut must hand off between two different actors (fromActorId !== toActorId)",
690
+ path: ["toActorId"]
691
+ });
692
+ var TransitionSchema = z7.union([TransitionCardSchema, TransitionCutSchema]);
529
693
  var ExplainSegmentPlanSchema = z7.object({
530
694
  id: z7.string().min(1),
531
695
  segmentPath: z7.string().regex(/^explain\/[a-z0-9][a-z0-9-]*\/segment\.mp4$/),
532
- durationMs: z7.number().int().positive()
696
+ durationMs: z7.number().int().positive(),
697
+ /**
698
+ * The hold on the capture segment *before* this scene, as on a cut or card joint
699
+ * (`TransitionCutSchema.freezeMs`). Only in plans with measured source bounds.
700
+ */
701
+ freezeMs: z7.number().int().positive().optional()
533
702
  });
534
703
  var HighlightRectSchema = z7.object({
535
704
  id: z7.string().min(1),
@@ -660,7 +829,23 @@ var SpeechSchema = z7.object({
660
829
  sampleRate: z7.literal(24e3),
661
830
  channels: z7.literal(1),
662
831
  clips: z7.array(SpeechClipSchema).min(1),
663
- engine: SpeechEngineSchema.optional()
832
+ engine: SpeechEngineSchema.optional(),
833
+ /**
834
+ * Normalise the assembled track's integrated loudness (ITU-R BS.1770) to `targetLufs`, with
835
+ * a make-up gain and a lookahead limiter that holds every sample at or under `maxPeakDbfs`.
836
+ * The short-form platforms play every upload at about -14 LUFS and turn a quieter one down
837
+ * no further — they only ever turn a louder one *down* — so an unnormalised narration simply
838
+ * plays quieter than everything around it.
839
+ *
840
+ * A target, not a gain: the gain and the limiting are a pure function of the frozen clips
841
+ * (`normalizeLoudness`, `@plaintake/audio`) recomputed by every freeze in a fixed sequence of
842
+ * double-precision operations, so the same clips always produce the same track. Absent means the clips are laid out at the level they
843
+ * were synthesised at, which is every plan before this field.
844
+ */
845
+ loudness: z7.object({
846
+ targetLufs: z7.number().min(-30).max(-5),
847
+ maxPeakDbfs: z7.number().min(-12).max(0)
848
+ }).optional()
664
849
  }).refine(
665
850
  ({ clips }) => new Set(clips.map((clip) => clip.id)).size === clips.length && clips.every(
666
851
  (clip, index) => index === 0 || clips[index - 1].atMs + clips[index - 1].durationMs <= clip.atMs
@@ -680,8 +865,8 @@ var SpeechSchema = z7.object({
680
865
  // A synthesised clip with no record of what synthesised it is the one state this block
681
866
  // exists to prevent. The converse is fine: an engine recorded on an all-file track is
682
867
  // merely redundant, not misleading. An `explain` clip is exempt rather than overlooked:
683
- // it was synthesised by plainmotion, whose identity the manifest's toolchain block
684
- // carries — the `engine` here would name a synthesiser that never touched it.
868
+ // its scene's own frozen plan records what synthesised it — the `engine` here names the
869
+ // step narrator's identity, which an explain scene's voice need not share.
685
870
  message: "engine must be recorded whenever any clip was synthesised",
686
871
  path: ["engine"]
687
872
  });
@@ -770,7 +955,31 @@ var RenderPlanSchema = z7.object({
770
955
  * `OutroSchema`'s colours are: the value reaches an FFmpeg filtergraph, and that
771
956
  * grammar admits no metacharacter.
772
957
  */
773
- padColor: HEX_COLOUR
958
+ padColor: HEX_COLOUR,
959
+ /**
960
+ * `follow` — the `--reframe follow` opt-in — fills the box with a 4:3 window of the
961
+ * capture that follows the camera shots (or a fixed centre window with no camera),
962
+ * instead of the whole 16:9 frame. The one place this plan crops the picture for its
963
+ * shape rather than for a zoom, and still a render-time decision: the window is a
964
+ * pure function of `camera.shots` and this box (`followWindow`, `renderer-ffmpeg`),
965
+ * so a bundle re-cuts into or out of it without re-recording. Absent means the whole
966
+ * frame is scaled in — every plan before this field, unchanged.
967
+ */
968
+ reframe: z7.literal("follow").optional(),
969
+ /**
970
+ * The two layouts that treat the picture as an object on a background rather than as
971
+ * the background itself — `--reframe social` and `--reframe inset`:
972
+ *
973
+ * - `social` (9:16, always `follow`): a 9:8 follow window in a box sized and placed for
974
+ * short-form feeds — clear of the platforms' own top bar, bottom caption block and
975
+ * right-hand action rail — with a band above it for the `hook`, bold captions below it,
976
+ * and rounded corners on a branded pad;
977
+ * - `inset` (16:9, never `follow`): the whole frame scaled into a box inset from every
978
+ * edge, with rounded corners on a branded pad.
979
+ *
980
+ * Absent means the plain letterbox or follow box — every plan before these, unchanged.
981
+ */
982
+ layout: z7.enum(["social", "inset"]).optional()
774
983
  }).optional(),
775
984
  /*
776
985
  * Everything before the closing card: the opening card, if there is one, plus the
@@ -801,6 +1010,21 @@ var RenderPlanSchema = z7.object({
801
1010
  message: "the letterbox box must fit inside the output frame",
802
1011
  path: ["letterbox"]
803
1012
  }
1013
+ ).refine(
1014
+ ({ letterbox }) => letterbox?.reframe === void 0 || (letterbox.layout === "social" ? letterbox.width * 8 === letterbox.height * 9 : letterbox.width * 3 === letterbox.height * 4),
1015
+ {
1016
+ // The follow window is cut to the box's own shape and scaled into it with no further
1017
+ // correction, so a box of any other shape would freeze a stretched picture: 4:3 for the
1018
+ // plain follow box, 9:8 for the social one.
1019
+ message: "a follow letterbox must be exactly 4:3 (9:8 under the social layout)",
1020
+ path: ["letterbox"]
1021
+ }
1022
+ ).refine(
1023
+ ({ letterbox }) => letterbox?.layout === void 0 || (letterbox.layout === "social" ? letterbox.reframe === "follow" : letterbox.reframe === void 0),
1024
+ {
1025
+ message: "the social layout always follows the camera and the inset layout never does",
1026
+ path: ["letterbox"]
1027
+ }
804
1028
  ),
805
1029
  captions: z7.object({
806
1030
  language: z7.string().min(2),
@@ -859,7 +1083,15 @@ var RenderPlanSchema = z7.object({
859
1083
  * remaining code, middle-centre (5), would put the words back over the picture — which
860
1084
  * is the thing the letterbox exists to stop.
861
1085
  */
862
- alignment: z7.union([z7.literal(2), z7.literal(8)]).default(2)
1086
+ alignment: z7.union([z7.literal(2), z7.literal(8)]).default(2),
1087
+ /**
1088
+ * The social layout's captions: bold, every character drawn by an explicitly chosen face
1089
+ * (`faceRuns`, `@plaintake/subtitles`, so an emoji or an arrow never depends on the font
1090
+ * provider's fallback), and — on a cue that carries `words` — each word lit in
1091
+ * `highlightColor` as it is spoken, the short-form convention. Absent means the caption style
1092
+ * every other plan has, so its ASS is byte-for-byte what it was.
1093
+ */
1094
+ social: z7.object({ highlightColor: HEX_COLOUR }).optional()
863
1095
  }),
864
1096
  ffmpeg: z7.object({
865
1097
  base: z7.array(z7.string()),
@@ -868,6 +1100,8 @@ var RenderPlanSchema = z7.object({
868
1100
  }),
869
1101
  intro: IntroSchema.optional(),
870
1102
  outro: OutroSchema.optional(),
1103
+ hook: HookSchema.optional(),
1104
+ frame: FrameSchema.optional(),
871
1105
  chapters: ChaptersSchema.optional(),
872
1106
  cursor: CursorSchema.optional(),
873
1107
  camera: CameraSchema.optional(),
@@ -929,8 +1163,8 @@ var RenderPlanSchema = z7.object({
929
1163
  * field of the five below that a segmented capture *always* carries; the other four are
930
1164
  * two independent, narrower capabilities layered on top of it:
931
1165
  *
932
- * - `actors`/`transitionCards`/`badgesAssPath` — the multi-actor demo's cast, the cards
933
- * drawn at each hand-off, and the badge track naming the active actor throughout. Three
1166
+ * - `actors`/`transitionCards`/`badgesAssPath` — the multi-actor demo's cast, the card
1167
+ * drawn (or the hard cut taken — `TransitionCutSchema`, zero screen time) at each hand-off, and the badge track naming the active actor throughout. Three
934
1168
  * fields that carry one capability and so, like every other optional block above, are
935
1169
  * present or absent together (the first refinement below) — and never without
936
1170
  * `segments`, which they reference by `actorId` (the second). Absent means no genuine
@@ -939,8 +1173,8 @@ var RenderPlanSchema = z7.object({
939
1173
  * whenever a scenario called `demo.explain()` at all, with or without a cast alongside
940
1174
  * it — never without `segments` either (the third refinement), for the same reason.
941
1175
  *
942
- * Every gap between adjacent `segments` is claimed by exactly one occupant, a card or an
943
- * explain scene, never both and never neither (the fifth refinement) — which is also why
1176
+ * Every gap between adjacent `segments` is claimed by exactly one occupant, a card, a cut or
1177
+ * an explain scene — a cut claims its gap with zero screen time, but it still claims it — never both and never neither (the fifth refinement) — which is also why
944
1178
  * `segments` can appear with `transitionCards` and `explainSegments` both absent only when
945
1179
  * `segments.length === 1` never occurs (`SegmentSchema`'s own `.min(2)` on the array rules
946
1180
  * it out): a lone, unsegmented capture has no gap to claim in the first place, and needs
@@ -948,7 +1182,7 @@ var RenderPlanSchema = z7.object({
948
1182
  */
949
1183
  actors: z7.array(ActorSchema).min(2).optional(),
950
1184
  segments: z7.array(SegmentSchema).min(2).optional(),
951
- transitionCards: z7.array(TransitionCardSchema).min(1).optional(),
1185
+ transitionCards: z7.array(TransitionSchema).min(1).optional(),
952
1186
  badgesAssPath: z7.literal("captions/badges.ass").optional(),
953
1187
  explainSegments: z7.array(ExplainSegmentPlanSchema).min(1).optional()
954
1188
  }).refine(
@@ -963,7 +1197,12 @@ var RenderPlanSchema = z7.object({
963
1197
  message: "actors, transitionCards and badgesAssPath must be present together or absent together",
964
1198
  path: ["actors"]
965
1199
  }
966
- ).refine((plan) => plan.actors === void 0 || plan.segments !== void 0, {
1200
+ ).refine((plan) => plan.frame === void 0 || plan.video.letterbox?.layout !== void 0, {
1201
+ // Corners are drawn on the box a layout lays the picture in; a plan with no such box has no
1202
+ // corner to round, and drawing one anyway would paint pad colour over the picture itself.
1203
+ message: "frame needs a social or inset letterbox to round the corners of",
1204
+ path: ["frame"]
1205
+ }).refine((plan) => plan.actors === void 0 || plan.segments !== void 0, {
967
1206
  // A cast with nowhere to draw its own windows — `segment.actorId` is what a badge or a
968
1207
  // transition card is drawn against — cannot render.
969
1208
  message: "actors requires segments \u2014 a cast with no capture windows of its own cannot render",
@@ -988,6 +1227,24 @@ var RenderPlanSchema = z7.object({
988
1227
  message: "segments must tile raw/session.webm contiguously from 0",
989
1228
  path: ["segments"]
990
1229
  }
1230
+ ).refine(
1231
+ (plan) => {
1232
+ if (plan.segments === void 0) return true;
1233
+ const { segments } = plan;
1234
+ const sourced = segments.filter(
1235
+ (segment) => segment.sourceStartMs !== void 0 || segment.sourceEndMs !== void 0
1236
+ );
1237
+ if (sourced.length === 0) return true;
1238
+ return segments.every(
1239
+ (segment, index) => segment.sourceStartMs !== void 0 && segment.sourceEndMs !== void 0 && segment.sourceEndMs > segment.sourceStartMs && (index === 0 || segment.sourceStartMs >= segments[index - 1].sourceEndMs)
1240
+ );
1241
+ },
1242
+ {
1243
+ // All or none: a splice that trimmed some segments by measured source position and the
1244
+ // rest by capture position would mix two coordinate systems in one graph.
1245
+ message: "sourceStartMs/sourceEndMs must be present on every segment or on none, each window non-empty and in order without overlap",
1246
+ path: ["segments"]
1247
+ }
991
1248
  ).refine(
992
1249
  (plan) => {
993
1250
  if (plan.segments === void 0) return true;
@@ -1100,7 +1357,20 @@ var ValidationResultSchema = z8.object({
1100
1357
  * without first trying to run it. Defaulted, so a result written before this parses.
1101
1358
  */
1102
1359
  handoff: z8.enum(["none", "preflight", "session"]).default("none"),
1103
- sha256: Sha2563
1360
+ sha256: Sha2563,
1361
+ /**
1362
+ * The terminal the scenario records instead of a web page, with its defaults applied, or
1363
+ * null for a browser scenario. Defaulted like `handoff`, so a result written before
1364
+ * terminals existed parses.
1365
+ */
1366
+ terminal: z8.object({
1367
+ cols: z8.number().int(),
1368
+ rows: z8.number().int(),
1369
+ shell: z8.string(),
1370
+ command: z8.array(z8.string()).optional(),
1371
+ /** Present (true) only when the scenario declares `terminal.browser`. */
1372
+ browser: z8.literal(true).optional()
1373
+ }).nullable().default(null)
1104
1374
  }).optional()
1105
1375
  });
1106
1376
  var RunCommandResultSchema = z8.object({
@@ -1381,21 +1651,16 @@ var DoctorResultSchema = z8.object({
1381
1651
  notes: z8.array(z8.string())
1382
1652
  }),
1383
1653
  /**
1384
- * Whether the plainmotion CLI is on `PATH`, for a scenario that wants to call
1385
- * `demo.explain()`. The same stance `hasAac`/`speech` above already take: absence is a
1386
- * *state*, not a fault — the overwhelming majority of scenarios never cut away to an
1387
- * explain scene, so `doctor` must not fail an otherwise-healthy installation over a binary
1388
- * most runs will never invoke. `assertPlainmotion` (`recorder-playwright`) is the same
1389
- * probe a run's own first `demo.explain()` pays, so `doctor` cannot say plainmotion is fine
1390
- * and then a run refuse it.
1654
+ * Whether a terminal scenario could be recorded right now: the vendored node-pty loads, and
1655
+ * xterm.js and the terminal font are in place. Never folded into `problems` — the same stance as
1656
+ * `speech` — because only a scenario declaring `terminal` needs any of it. Optional so a result
1657
+ * written before terminals existed parses.
1391
1658
  */
1392
- plainmotion: z8.object({
1393
- installed: z8.boolean(),
1394
- /** plainmotion's own `--version` output. Absent when not installed. */
1395
- version: z8.string().optional(),
1396
- /** Why the probe failed, in the same words a run's own refusal would use. Absent when installed. */
1397
- note: z8.string().optional()
1398
- })
1659
+ terminal: z8.object({
1660
+ ready: z8.boolean(),
1661
+ /** What is missing, in the words a refused run would use. Empty when ready. */
1662
+ notes: z8.array(z8.string())
1663
+ }).optional()
1399
1664
  });
1400
1665
  var okMatchesProblems = (value) => value.ok === (value.problems.length === 0);
1401
1666
  var ActivateReasonSchema = z8.enum([
@@ -1451,6 +1716,7 @@ var PublishReasonSchema = z8.enum([
1451
1716
  "hash-mismatch",
1452
1717
  "length-mismatch",
1453
1718
  "conflict",
1719
+ "taken-down",
1454
1720
  "unexpected-response"
1455
1721
  ]);
1456
1722
  var PublishResultSchema = z8.discriminatedUnion("ok", [
@@ -1467,6 +1733,12 @@ var PublishResultSchema = z8.discriminatedUnion("ok", [
1467
1733
  url: z8.string().regex(/^https?:\/\/\S+$/, "must be an http or https URL"),
1468
1734
  /** False when the video was already on the service — nothing was uploaded. */
1469
1735
  uploaded: z8.boolean(),
1736
+ /**
1737
+ * True when the share had been unpublished (or had expired) and this publish brought
1738
+ * it back — `unpublish`'s undo. Defaults to false so a result written before the
1739
+ * field existed still parses.
1740
+ */
1741
+ restored: z8.boolean().default(false),
1470
1742
  /**
1471
1743
  * Present only when `--remember` was asked: true when the key was stored beside
1472
1744
  * license.json, false when the key file could not be written. The service's
@@ -1486,6 +1758,42 @@ var PublishResultSchema = z8.discriminatedUnion("ok", [
1486
1758
  poster: z8.boolean()
1487
1759
  })
1488
1760
  ]).refine(okMatchesProblems, "ok must be true iff problems is empty");
1761
+ var UnpublishReasonSchema = z8.enum([
1762
+ "no-endpoint",
1763
+ "no-key",
1764
+ "invalid-store",
1765
+ "not-a-bundle",
1766
+ "invalid-target",
1767
+ "endpoint-mismatch",
1768
+ "unauthorized",
1769
+ "forbidden",
1770
+ "not-found",
1771
+ "unreachable",
1772
+ "unexpected-response"
1773
+ ]);
1774
+ var UnpublishResultSchema = z8.discriminatedUnion("ok", [
1775
+ z8.object({
1776
+ ...envelope("unpublish"),
1777
+ ok: z8.literal(false),
1778
+ reason: UnpublishReasonSchema
1779
+ }),
1780
+ z8.object({
1781
+ ...envelope("unpublish"),
1782
+ ok: z8.literal(true),
1783
+ videoId: z8.string().regex(/^[a-z2-7]{26}$/, "must be a 26-char base32 share id"),
1784
+ /** The page that now answers 410. */
1785
+ url: z8.string().regex(/^https?:\/\/\S+$/, "must be an http or https URL"),
1786
+ /** When the share went dark; repeating an unpublish keeps the first instant. */
1787
+ deletedAt: z8.string().min(1),
1788
+ /**
1789
+ * Who holds the tombstone. `admin` means the service's operator had already taken
1790
+ * the share down — still a success (it is dark), but only the operator can undo it.
1791
+ */
1792
+ deletedBy: z8.enum(["producer", "admin", "expired"]),
1793
+ /** When retention may delete the bytes for good; a republish before then restores. */
1794
+ purgeAfter: z8.string().min(1)
1795
+ })
1796
+ ]).refine(okMatchesProblems, "ok must be true iff problems is empty");
1489
1797
 
1490
1798
  // ../schema/src/zod-format.ts
1491
1799
  import { ZodError } from "zod";
@@ -26,6 +26,7 @@ export declare const BundleManifestSchema: z.ZodObject<{
26
26
  timezone: z.ZodLiteral<"UTC">;
27
27
  viewport: z.ZodTuple<[z.ZodLiteral<1920>, z.ZodLiteral<1080>], null>;
28
28
  deviceScaleFactor: z.ZodLiteral<1>;
29
+ uiScale: z.ZodOptional<z.ZodUnion<readonly [z.ZodLiteral<1.5>, z.ZodLiteral<2>]>>;
29
30
  }, z.core.$strip>;
30
31
  assertions: z.ZodArray<z.ZodObject<{
31
32
  id: z.ZodString;
@@ -19,7 +19,7 @@ import { z } from 'zod';
19
19
  * version): those come from the current build, never from the file.
20
20
  *
21
21
  * Only `synth`-sourced clips appear. A `--speech file` clip is the author's own audio with nothing
22
- * to synthesise, and an `explain` scene is plainmotion's, not this narrator's.
22
+ * to synthesise, and an `explain` scene's narration is synthesised by its own compile, not this narrator.
23
23
  */
24
24
  export declare const NARRATION_INDEX_SCHEMA = "agent-demo.narration-index/v1";
25
25
  /**