@plaintake/scenario 1.3.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -79,6 +79,9 @@ var ScenarioCameraSchema = z2.object({
79
79
  easeMs: z2.number().int().min(133, "camera.easeMs must be at least 133ms").max(2e3, "camera.easeMs must be at most 2000ms").optional(),
80
80
  minDwellMs: z2.number().int().min(0, "camera.minDwellMs must be at least 0").max(5e3, "camera.minDwellMs must be at most 5000ms").optional()
81
81
  });
82
+ var ScenarioSpeechSchema = z2.object({
83
+ speed: z2.number().min(0.5, "speech.speed must be at least 0.5 (half speed)").max(2, "speech.speed must be at most 2.0 (double speed) \u2014 an unmeasured, conservative ceiling").optional()
84
+ });
82
85
  var ScenarioMetaSchema = z2.object({
83
86
  schema: z2.literal("agent-demo.scenario/v1"),
84
87
  id: z2.string().regex(/^[a-z0-9][a-z0-9-]*$/, "must be a lowercase kebab-case id"),
@@ -142,7 +145,34 @@ var ScenarioMetaSchema = z2.object({
142
145
  * Note this is read only when the camera is on at all (`--camera zoom`). A scenario
143
146
  * recorded flat carries these harmlessly.
144
147
  */
145
- camera: ScenarioCameraSchema.optional()
148
+ camera: ScenarioCameraSchema.optional(),
149
+ /*
150
+ * Scenario-level narration speed. Optional and inert when absent, the same precedent as
151
+ * `camera` above: a scenario that says nothing about speed narrates at 1x, exactly as every
152
+ * scenario written before this field existed always has. See `ScenarioSpeechSchema`'s own
153
+ * comment for the field's shape and the reasoning behind `speed`'s bounds.
154
+ */
155
+ speech: ScenarioSpeechSchema.optional(),
156
+ /*
157
+ * A text-substitution dictionary applied to narration immediately before synthesis — not the
158
+ * vendored en-us phoneme dictionary `install-voice` fetches (a different thing entirely: that
159
+ * one is a fixed grapheme-to-phoneme table the engine consults for every word; this one is
160
+ * authored per scenario, substitutes whole words, and is applied by `narrate.ts` before the
161
+ * text ever reaches the engine). `{ SQL: 'sequel' }` makes the voice say "sequel" while the
162
+ * caption keeps reading "SQL" — captions read from `step.subtitle`/recorded events, an
163
+ * entirely separate pipeline this dictionary never touches.
164
+ *
165
+ * Keys must be non-empty, so an accidental `{ '': 'x' }` is refused here rather than silently
166
+ * matching nothing at synthesis time. Values may be empty: an empty value mutes the matched
167
+ * word out of the spoken line, which is a deliberate if narrow use — nothing else in the DSL
168
+ * offers a way to drop a word from the audio without also dropping it from the caption.
169
+ * Word-boundary, case-insensitive matching; see `applyPronunciations` in
170
+ * `recorder-playwright/src/narrate.ts`, the one place this dictionary is actually applied,
171
+ * for the exact matching rule.
172
+ *
173
+ * Optional and inert when absent, the same precedent as everything else on this schema.
174
+ */
175
+ pronunciations: z2.record(z2.string().min(1, "pronunciation keys must not be empty"), z2.string()).optional()
146
176
  });
147
177
 
148
178
  // ../schema/src/manifest.ts
@@ -234,12 +264,36 @@ var CursorPointSchema = z5.object({
234
264
  action: z5.enum(["click", "type", "point"]),
235
265
  rippleMs: z5.number().int().nonnegative().optional(),
236
266
  /**
237
- * Counter-scale percent, so the cursor stays one size on screen while the camera
238
- * magnifies its window — up to 1.6x — and the emitter compensates with `\fscx`/`\fscy`.
239
- * Absent means 100, so it carries no default: a cursor bundle from before the camera
240
- * existed still parses and regenerates byte-for-byte.
267
+ * Size-correction percent, applied by the emitter with `\fscx`/`\fscy`, so the arrow
268
+ * stays one size on screen whatever the frame does to it between the cursor filter and
269
+ * the output. Two things do, and they compose into one number:
270
+ *
271
+ * - **The camera magnifies it.** The cursor is drawn before the crop, so a zoomed shot
272
+ * enlarges the pointer with the page — up to 1.6x. That alone only ever *shrinks* the
273
+ * correction, which is why this field's ceiling was 100 for as long as the camera was
274
+ * the only thing acting on it.
275
+ * - **The letterbox shrinks it.** Under `--aspect`, the finished 16:9 frame is scaled
276
+ * down onto `video.letterbox` before being padded into the taller output, and the
277
+ * cursor `ass` runs upstream of that scale — so a 12x19-unit arrow arrives at the
278
+ * viewer smaller than it was drawn, which is the exact failure this field exists to
279
+ * prevent. The correction is the inverse of that scale, and it telescopes with the
280
+ * camera's into a single ratio of the active window to the letterbox width:
281
+ *
282
+ * ```
283
+ * scale% = round(100 * active.w / letterbox.width)
284
+ * ```
285
+ *
286
+ * A 9:16 full-frame shot is `round(100 * 1920 / 1080)` = **178**, so the old `max(100)`
287
+ * rejected the corrected value outright. 400 is the ceiling now: not a computed bound —
288
+ * the shipped geometry tops out at 178 — but a loose one that still catches a derivation
289
+ * off by an order of magnitude rather than freezing it into a plan.
290
+ *
291
+ * Absent means 100, so it carries no default. A 16:9 plan has no letterbox and computes
292
+ * `100 * 1920 / 1920` = 100, which the emitter's omit-defaults rule drops exactly as it
293
+ * always did — so a cursor bundle from before the camera, or from before the letterbox,
294
+ * still parses and regenerates byte-for-byte.
241
295
  */
242
- scale: z5.number().int().min(1).max(100).optional()
296
+ scale: z5.number().int().min(1).max(400).optional()
243
297
  });
244
298
  var CursorSchema = z5.object({
245
299
  assPath: z5.literal("captions/cursor.ass"),
@@ -295,6 +349,35 @@ var CameraSchema = z5.object({
295
349
  path: ["shots"]
296
350
  }
297
351
  );
352
+ var HighlightRectSchema = z5.object({
353
+ id: z5.string().min(1),
354
+ x: z5.number().int().min(0),
355
+ y: z5.number().int().min(0),
356
+ width: z5.number().int().positive(),
357
+ height: z5.number().int().positive(),
358
+ startMs: z5.number().int().nonnegative(),
359
+ endMs: z5.number().int().positive(),
360
+ label: z5.string().min(1).optional()
361
+ }).refine(({ x, y, width, height }) => x + width <= 1920 && y + height <= 1080, {
362
+ // Unlike CameraShotSchema's crop window, a highlight rect is drawn by libass, not fed to
363
+ // `crop` — there is no yuv420p even-pixel rule to enforce — but it still has to fit the
364
+ // 1920x1080 canvas the spotlight is cut from, or the hole would miss the frame it is
365
+ // meant to sit inside.
366
+ message: "the rect must stay inside the 1920x1080 frame",
367
+ path: ["x"]
368
+ });
369
+ var HighlightSchema = z5.object({
370
+ assPath: z5.literal("captions/highlight.ass"),
371
+ rects: z5.array(HighlightRectSchema).min(1)
372
+ }).refine(
373
+ ({ rects }) => rects.every(
374
+ (rect, index) => rect.endMs > rect.startMs && (index === 0 || rects[index - 1].endMs <= rect.startMs)
375
+ ),
376
+ {
377
+ message: "rects must be ordered, non-overlapping, and end after they start",
378
+ path: ["rects"]
379
+ }
380
+ );
298
381
  var SpeechClipSchema = z5.object({
299
382
  id: z5.string().min(1),
300
383
  /*
@@ -364,11 +447,84 @@ var RenderPlanSchema = z5.object({
364
447
  sha256: z5.string().regex(/^[0-9a-f]{64}$/),
365
448
  durationMs: z5.number().int().nonnegative()
366
449
  }),
450
+ /**
451
+ * The frame. **Three geometries live here and they are not the same thing**, which is the
452
+ * whole reason this block stopped being two literals:
453
+ *
454
+ * - `capture` is the crop space — the fixed 1920x1080 Chromium always records at, DPR 1,
455
+ * which never varies and is not allowed to (`AGENTS.md` §1 constraint 3). Everything
456
+ * upstream of the output is expressed in it: cursor points, camera windows, highlight
457
+ * rects. It is `.default()`ed
458
+ * rather than required so a plan frozen before the field existed parses to exactly the
459
+ * space it was rendered in, and the arguments derived from it come out byte-identical.
460
+ * - `width`/`height` are the **output** frame: 1920x1080 for 16:9, 1080x1920 for 9:16,
461
+ * 1080x1080 for 1:1. They are no longer literals because that is the one number
462
+ * `--aspect` moves.
463
+ * - `letterbox` is where the finished 16:9 picture is laid inside that output, plus the
464
+ * colour filling the rest. Absent on 16:9, where output and capture coincide and there
465
+ * is nothing to pad.
466
+ *
467
+ * **It letterboxes; it does not crop.** A true 9:16 crop of a 1920x1080 capture is 607px
468
+ * wide and throws away 68% of the UI, which is useless for a product demo. Scaling the
469
+ * finished frame down and padding it into the taller output keeps every pixel, puts the
470
+ * captions in the empty band instead of over the video, and — because none of it touches
471
+ * capture — makes a bundle already recorded re-cuttable to vertical without re-recording, the
472
+ * same property soft subtitles have. The one exception is the captions themselves: a `Cue`
473
+ * carries the lines the planner already *wrapped*, so a bundle whose lines were wrapped for a
474
+ * wider frame is refused by `recutPlan` rather than re-cut into text that overruns the band.
475
+ *
476
+ * **`letterbox` is one object, not a box and a colour side by side.** Every other optional
477
+ * capability in this plan — `intro`, `outro`, `chapters`, `cursor`, `camera`, `highlight`,
478
+ * `speech` — is a single optional object carrying everything it needs, and this follows
479
+ * them. Two independent optional siblings would have made "a box with no fill" and "a fill
480
+ * with no box" representable in the *type*, leaving a runtime refinement as the only thing
481
+ * standing between a derivation and a plan nothing can render; as one object the invalid
482
+ * pair is unconstructable and `tsc` says so at every call site.
483
+ *
484
+ * **It is deliberately not called a "plate".** That word is taken: the dark box behind the
485
+ * captions is the plate everywhere else in this codebase — `AGENTS.md`, `probe.ts`,
486
+ * `args.ts`, the hard-subtitle acceptance thresholds — and the two would sit within a few
487
+ * lines of each other in the filtergraph. `/v1` is frozen for good (capabilities are
488
+ * additive fields, never a version bump), so a wire name here is effectively permanent and
489
+ * worth spending the thought on now rather than after two more slices read it.
490
+ *
491
+ * **The box is 1080x608 and that is a measured, accepted 0.08% stretch.** Scaling
492
+ * 1920x1080 to `w=1080` gives `h = 607.5`, which is neither integer nor even. 608 is taken
493
+ * instead, stretching the picture vertically by `608/607.5` = 1.00082 — 0.08%, well under
494
+ * the ~1% at which a stretch becomes visible on a face or a circle, and this tool records
495
+ * web UIs. The exact-16:9 alternative, 1056x594, is **rejected**: it is correct on paper
496
+ * and wrong on screen, because a 1056-wide box in a 1080-wide output leaves a 12px
497
+ * pillarbox bar down each side — bars inside a design whose entire purpose is to put the
498
+ * bars somewhere useful.
499
+ *
500
+ * **Every dimension here is even, and the two halves fail differently.** On `yuv420p` the
501
+ * chroma planes are half-resolution, so an odd `letterbox` `x`/`y`/`width`/`height` is
502
+ * silently truncated down a pixel by FFmpeg rather than refused — the invisible 1px
503
+ * misalignment `CameraSchema` cites (M10) for its own crop window, and the reason it keeps
504
+ * an evenness check on `w`/`h` that `%32` and `9w/16` already imply. An odd **output**
505
+ * `width`/`height` is the opposite: `libx264` hard-fails the encode with `width not
506
+ * divisible by 2`. That is the loud case, and leaving it unguarded while guarding the quiet
507
+ * one would be backwards — a plan is frozen once and rendered later, so both belong at the
508
+ * parse, not at the encode.
509
+ */
367
510
  video: z5.object({
368
- width: z5.literal(1920),
369
- height: z5.literal(1080),
511
+ width: z5.number().int().positive().multipleOf(2),
512
+ height: z5.number().int().positive().multipleOf(2),
370
513
  fps: z5.literal(30),
371
514
  pixelFormat: z5.literal("yuv420p"),
515
+ capture: z5.object({ width: z5.literal(1920), height: z5.literal(1080) }).default({ width: 1920, height: 1080 }),
516
+ letterbox: z5.object({
517
+ width: z5.number().int().positive().multipleOf(2),
518
+ height: z5.number().int().positive().multipleOf(2),
519
+ x: z5.number().int().nonnegative().multipleOf(2),
520
+ y: z5.number().int().nonnegative().multipleOf(2),
521
+ /**
522
+ * The fill for everything the box does not cover. `#RRGGBB` for the reason
523
+ * `OutroSchema`'s colours are: the value reaches an FFmpeg filtergraph, and that
524
+ * grammar admits no metacharacter.
525
+ */
526
+ padColor: HEX_COLOUR
527
+ }).optional(),
372
528
  /*
373
529
  * Everything before the closing card: the opening card, if there is one, plus the
374
530
  * recorded content. Not the content alone.
@@ -386,7 +542,19 @@ var RenderPlanSchema = z5.object({
386
542
  */
387
543
  durationMs: z5.number().int().positive(),
388
544
  tailPadMs: z5.number().int().nonnegative()
389
- }),
545
+ }).refine(
546
+ ({ width, height, letterbox }) => letterbox === void 0 || letterbox.x + letterbox.width <= width && letterbox.y + letterbox.height <= height,
547
+ {
548
+ // The same containment `CameraShotSchema` enforces against the 1920x1080 capture and
549
+ // `HighlightRectSchema` against the canvas it cuts a hole in — except the bound here
550
+ // is a sibling field rather than a constant, which is exactly why it has to live on
551
+ // `video` and not on `letterbox` itself. A box hanging off the frame is not a
552
+ // rendering FFmpeg refuses; `pad` quietly clips it, so the plan would freeze a
553
+ // picture nobody ever sees the edge of.
554
+ message: "the letterbox box must fit inside the output frame",
555
+ path: ["letterbox"]
556
+ }
557
+ ),
390
558
  captions: z5.object({
391
559
  language: z5.string().min(2),
392
560
  srtPath: z5.literal("captions/captions.srt"),
@@ -425,7 +593,26 @@ var RenderPlanSchema = z5.object({
425
593
  * and ASS's own default, so a plan frozen before this field existed regenerates its ASS
426
594
  * byte-for-byte rather than byte-for-byte-except-one-field-nobody-can-see.
427
595
  */
428
- boxOpacity: z5.number().min(0).max(1).default(1)
596
+ boxOpacity: z5.number().min(0).max(1).default(1),
597
+ /**
598
+ * ASS `\an` alignment: which edge of the frame the caption block is anchored to, and
599
+ * therefore which margin `marginBottom` is measured from.
600
+ *
601
+ * `2` is bottom-centre — the only thing captions have ever done here, and the default,
602
+ * so a plan frozen before this field regenerates its ASS byte-for-byte. `8` is
603
+ * top-centre, and it exists for the letterbox: a 9:16 picture sits **high** in the output
604
+ * rather than centred, because a centred one leaves a 656px bottom bar and puts a
605
+ * bottom-anchored caption 600px from the content it describes, which reads as two
606
+ * unrelated things. Anchoring the band to the top of the pad instead keeps the words
607
+ * next to the picture they explain.
608
+ *
609
+ * The union is closed to those two rather than opened to all nine `\an` codes: a
610
+ * caption band spans the full width by construction, so every left- or right-anchored
611
+ * code (1, 3, 4, 6, 7, 9) describes a band this renderer cannot produce, and the one
612
+ * remaining code, middle-centre (5), would put the words back over the picture — which
613
+ * is the thing the letterbox exists to stop.
614
+ */
615
+ alignment: z5.union([z5.literal(2), z5.literal(8)]).default(2)
429
616
  }),
430
617
  ffmpeg: z5.object({
431
618
  base: z5.array(z5.string()),
@@ -437,6 +624,7 @@ var RenderPlanSchema = z5.object({
437
624
  chapters: ChaptersSchema.optional(),
438
625
  cursor: CursorSchema.optional(),
439
626
  camera: CameraSchema.optional(),
627
+ highlight: HighlightSchema.optional(),
440
628
  speech: SpeechSchema.optional()
441
629
  });
442
630
 
@@ -543,6 +731,19 @@ var RunCommandResultSchema = z6.object({
543
731
  assertions: z6.array(AssertionSchema),
544
732
  outputs: z6.array(ArtifactRefSchema)
545
733
  });
734
+ var CheckAssertionSchema = AssertionSchema.extend({ message: z6.string().optional() });
735
+ var CheckCommandResultSchema = z6.object({
736
+ ...envelope("check"),
737
+ bundleDir: DISPLAY_PATH,
738
+ status: StatusSchema,
739
+ assertions: z6.array(CheckAssertionSchema),
740
+ /**
741
+ * Handoff notes only. Cue, chapter, cursor and camera diagnostics are all derived from a
742
+ * render plan that `check` never builds — fabricating them would report on capabilities
743
+ * this command cannot see.
744
+ */
745
+ diagnostics: z6.array(z6.object({ code: z6.string(), cueId: z6.string(), detail: z6.string() }))
746
+ });
546
747
  var RenderCommandResultSchema = z6.object({
547
748
  ...envelope("render"),
548
749
  bundleDir: DISPLAY_PATH,
@@ -559,6 +760,33 @@ var VerificationReportSchema = z6.object({
559
760
  bundleDir: DISPLAY_PATH,
560
761
  artifactCount: z6.number().int().nonnegative()
561
762
  });
763
+ var DifferenceCategorySchema = z6.enum(["step", "assertion", "target", "timing", "caption"]);
764
+ var DiffCommandResultSchema = z6.object({
765
+ ...envelope("diff"),
766
+ bundleA: DISPLAY_PATH,
767
+ bundleB: DISPLAY_PATH,
768
+ identical: z6.boolean(),
769
+ differences: z6.array(z6.object({ category: DifferenceCategorySchema, detail: z6.string() }))
770
+ });
771
+ var PruneCandidateSchema = z6.object({
772
+ path: DISPLAY_PATH,
773
+ bytes: z6.number().int().nonnegative(),
774
+ scenarioId: z6.string().optional()
775
+ });
776
+ var PruneCommandResultSchema = z6.object({
777
+ ...envelope("prune"),
778
+ dryRun: z6.boolean(),
779
+ /** What was removed (confirmed) or would be (dry run) — never includes a `failed` entry. */
780
+ candidates: z6.array(PruneCandidateSchema),
781
+ /** A discovered bundle that was never attempted, and why — never silently dropped. */
782
+ skipped: z6.array(z6.object({ path: DISPLAY_PATH, reason: z6.string() })),
783
+ /** A selected candidate `rm` was attempted on and could not remove, and why. Always empty
784
+ * on a dry run, since nothing is attempted until `--yes`. */
785
+ failed: z6.array(z6.object({ path: DISPLAY_PATH, reason: z6.string() })),
786
+ /** What `candidates` account for: bytes actually reclaimed if `dryRun` is false, or would
787
+ * be reclaimed if every candidate succeeded. */
788
+ bytesReclaimed: z6.number().int().nonnegative()
789
+ });
562
790
  var InspectResultSchema = z6.object({
563
791
  ...envelope("inspect"),
564
792
  bundleDir: DISPLAY_PATH,
@@ -85,8 +85,18 @@ export declare const ChaptersSchema: z.ZodObject<{
85
85
  export type Chapters = z.infer<typeof ChaptersSchema>;
86
86
  export type ChapterMark = z.infer<typeof ChapterMarkSchema>;
87
87
  /**
88
- * One point on the cursor track. `x`/`y` are the centre of the step's `target`
89
- * rect, in PlayRes (= viewport) pixels, so the same coordinates libass will draw at.
88
+ * One point on the cursor track. `x`/`y` are the centre of the step's `target` rect, in capture
89
+ * pixels — the 1920x1080 space the cursor ASS declares as its own PlayRes and the space the
90
+ * camera derivation reads these same numbers in.
91
+ *
92
+ * They are the **measured position**, not necessarily the emitted one. `toCursorAss` pulls the
93
+ * anchor back where the arrow's body would otherwise leave the frame, and how far it has to pull
94
+ * depends on `scale` — which depends on the output shape, and so is not a property of the
95
+ * recording at all. Keeping the containment fix at the drawing rather than in the frozen value is
96
+ * what makes `recutPlan` a lossless round trip: re-cutting to `9:16` and back reproduces the
97
+ * point it started from, where an in-place clamp moved an edge-flush anchor and could not move it
98
+ * back.
99
+ *
90
100
  * `arriveMs` is when the cursor reaches the point — no later than the step's start, and
91
101
  * pulled earlier for clicks so the arrow is parked and visible before it ripples — and
92
102
  * `departMs` is when it may leave for the next point, pulled back from the step's finish
@@ -180,6 +190,52 @@ export declare const CameraSchema: z.ZodObject<{
180
190
  }, z.core.$strip>;
181
191
  export type Camera = z.infer<typeof CameraSchema>;
182
192
  export type CameraShot = z.infer<typeof CameraShotSchema>;
193
+ /**
194
+ * One dimmed-frame-with-a-hole window: a step's measured target rect, frozen exactly as
195
+ * `CursorPointSchema` freezes a target's centre, and for the same reason. `renderer-ffmpeg`'s
196
+ * `highlightTrackFromEvents` re-reading `events/events.ndjson` at render time would violate
197
+ * the invariant every other derived render file already obeys — `captions/highlight.ass` has
198
+ * to be a pure function of the plan, so the rect a step measured has to survive plan freeze
199
+ * as a rect, not merely as a decision to draw one.
200
+ *
201
+ * `startMs`/`endMs` bound the whole step, not a retimed lead/rest/glide the way a cursor
202
+ * point is — a spotlight has no motion to choreograph, so unlike `CursorPointSchema` there is
203
+ * nothing here to pull earlier or hold open.
204
+ *
205
+ * `label` is the optional callout text a scenario may attach (`highlight: { label: '...' }`).
206
+ * Absent means the spotlight alone, exactly what `highlight: true` (no object) asks for.
207
+ */
208
+ export declare const HighlightRectSchema: z.ZodObject<{
209
+ id: z.ZodString;
210
+ x: z.ZodNumber;
211
+ y: z.ZodNumber;
212
+ width: z.ZodNumber;
213
+ height: z.ZodNumber;
214
+ startMs: z.ZodNumber;
215
+ endMs: z.ZodNumber;
216
+ label: z.ZodOptional<z.ZodString>;
217
+ }, z.core.$strip>;
218
+ /**
219
+ * The highlight track. A rect **list**, ordered and non-overlapping on the same discipline
220
+ * `CursorSchema`'s points already enforce — the derivation emits one entry per highlighted
221
+ * step, in event order, so an overlap or an out-of-order entry can only mean the derivation
222
+ * is broken.
223
+ */
224
+ export declare const HighlightSchema: z.ZodObject<{
225
+ assPath: z.ZodLiteral<"captions/highlight.ass">;
226
+ rects: z.ZodArray<z.ZodObject<{
227
+ id: z.ZodString;
228
+ x: z.ZodNumber;
229
+ y: z.ZodNumber;
230
+ width: z.ZodNumber;
231
+ height: z.ZodNumber;
232
+ startMs: z.ZodNumber;
233
+ endMs: z.ZodNumber;
234
+ label: z.ZodOptional<z.ZodString>;
235
+ }, z.core.$strip>>;
236
+ }, z.core.$strip>;
237
+ export type Highlight = z.infer<typeof HighlightSchema>;
238
+ export type HighlightRect = z.infer<typeof HighlightRectSchema>;
183
239
  /**
184
240
  * One frozen narration clip and where it lands on the timeline.
185
241
  *
@@ -271,6 +327,9 @@ export type ChaptersInput = {
271
327
  /** The camera decision as the application layer supplies it; `commandPath` is filled in
272
328
  * by `buildRenderPlan`, exactly as `CursorInput` and `ChaptersInput` work. */
273
329
  export type CameraInput = Omit<Camera, 'commandPath'>;
330
+ /** The highlight decision as the application layer supplies it; `assPath` is filled in by
331
+ * `buildRenderPlan`, exactly as `CursorInput` works. */
332
+ export type HighlightInput = Omit<Highlight, 'assPath'>;
274
333
  /**
275
334
  * The branding decision as the application layer supplies it: everything about the card
276
335
  * except where its ASS file lives, which `buildRenderPlan` fills in. Declared here rather
@@ -327,9 +386,19 @@ export declare const RENDER_PLAN_SCHEMA = "agent-demo.render/v1";
327
386
  * fails at the parse, naming the value it should have — which is a better outcome than the
328
387
  * ladder's, where the same bundle failed while carrying arguments that would have rendered.
329
388
  *
330
- * A plan using no optional capability remains byte-identical to what every earlier version
331
- * produced, so `scripts/make-golden-bundle.sh` regenerates the committed golden bundle
332
- * exactly — the version string was already `/v1` there.
389
+ * A plan using no optional capability still executes byte-identically to what every earlier
390
+ * version produced — the version string was already `/v1` there, and more to the point
391
+ * `ffmpeg.base`/`soft`/`hard` are unchanged, so the `demo.mp4` a re-render yields is the same
392
+ * file down to the byte.
393
+ *
394
+ * **The plan's own JSON is a weaker claim, and deliberately so.** Adding a field with a Zod
395
+ * `.default()` materialises it on parse, so `stableStringify` writes it and
396
+ * `render/render-plan.json` gains bytes the committed golden fixture does not have — which is
397
+ * what `style`'s defaults (`marginSide`, `boxOpacity`, …) already did once, and what
398
+ * `video.capture` and `style.alignment` do now. That makes the *plan file* differ from the
399
+ * committed golden until `scripts/make-golden-bundle.sh` is re-run; it does not touch the
400
+ * rendered output, because nothing derived from those defaults reaches an argument array. The
401
+ * reproducibility claim this project actually makes is about the video, and it holds.
333
402
  */
334
403
  export declare const RenderPlanSchema: z.ZodObject<{
335
404
  schema: z.ZodLiteral<"agent-demo.render/v1">;
@@ -339,10 +408,21 @@ export declare const RenderPlanSchema: z.ZodObject<{
339
408
  durationMs: z.ZodNumber;
340
409
  }, z.core.$strip>;
341
410
  video: z.ZodObject<{
342
- width: z.ZodLiteral<1920>;
343
- height: z.ZodLiteral<1080>;
411
+ width: z.ZodNumber;
412
+ height: z.ZodNumber;
344
413
  fps: z.ZodLiteral<30>;
345
414
  pixelFormat: z.ZodLiteral<"yuv420p">;
415
+ capture: z.ZodDefault<z.ZodObject<{
416
+ width: z.ZodLiteral<1920>;
417
+ height: z.ZodLiteral<1080>;
418
+ }, z.core.$strip>>;
419
+ letterbox: z.ZodOptional<z.ZodObject<{
420
+ width: z.ZodNumber;
421
+ height: z.ZodNumber;
422
+ x: z.ZodNumber;
423
+ y: z.ZodNumber;
424
+ padColor: z.ZodString;
425
+ }, z.core.$strip>>;
346
426
  durationMs: z.ZodNumber;
347
427
  tailPadMs: z.ZodNumber;
348
428
  }, z.core.$strip>;
@@ -371,6 +451,7 @@ export declare const RenderPlanSchema: z.ZodObject<{
371
451
  outlineOpacity: z.ZodDefault<z.ZodNumber>;
372
452
  boxColor: z.ZodDefault<z.ZodString>;
373
453
  boxOpacity: z.ZodDefault<z.ZodNumber>;
454
+ alignment: z.ZodDefault<z.ZodUnion<readonly [z.ZodLiteral<2>, z.ZodLiteral<8>]>>;
374
455
  }, z.core.$strip>;
375
456
  ffmpeg: z.ZodObject<{
376
457
  base: z.ZodArray<z.ZodString>;
@@ -428,6 +509,19 @@ export declare const RenderPlanSchema: z.ZodObject<{
428
509
  h: z.ZodNumber;
429
510
  }, z.core.$strip>>;
430
511
  }, z.core.$strip>>;
512
+ highlight: z.ZodOptional<z.ZodObject<{
513
+ assPath: z.ZodLiteral<"captions/highlight.ass">;
514
+ rects: z.ZodArray<z.ZodObject<{
515
+ id: z.ZodString;
516
+ x: z.ZodNumber;
517
+ y: z.ZodNumber;
518
+ width: z.ZodNumber;
519
+ height: z.ZodNumber;
520
+ startMs: z.ZodNumber;
521
+ endMs: z.ZodNumber;
522
+ label: z.ZodOptional<z.ZodString>;
523
+ }, z.core.$strip>>;
524
+ }, z.core.$strip>>;
431
525
  speech: z.ZodOptional<z.ZodObject<{
432
526
  trackPath: z.ZodLiteral<"speech/narration.wav">;
433
527
  sampleRate: z.ZodLiteral<24000>;
@@ -52,6 +52,30 @@ export declare const RunCommandResultSchema: z.ZodObject<{
52
52
  ok: z.ZodBoolean;
53
53
  problems: z.ZodArray<z.ZodString>;
54
54
  }, z.core.$strip>;
55
+ export declare const CheckCommandResultSchema: z.ZodObject<{
56
+ bundleDir: z.ZodString;
57
+ status: z.ZodEnum<{
58
+ passed: "passed";
59
+ failed: "failed";
60
+ }>;
61
+ assertions: z.ZodArray<z.ZodObject<{
62
+ id: z.ZodString;
63
+ status: z.ZodEnum<{
64
+ passed: "passed";
65
+ failed: "failed";
66
+ }>;
67
+ message: z.ZodOptional<z.ZodString>;
68
+ }, z.core.$strip>>;
69
+ diagnostics: z.ZodArray<z.ZodObject<{
70
+ code: z.ZodString;
71
+ cueId: z.ZodString;
72
+ detail: z.ZodString;
73
+ }, z.core.$strip>>;
74
+ schema: z.ZodLiteral<"agent-demo.result/v1">;
75
+ kind: z.ZodLiteral<"check">;
76
+ ok: z.ZodBoolean;
77
+ problems: z.ZodArray<z.ZodString>;
78
+ }, z.core.$strip>;
55
79
  export declare const RenderCommandResultSchema: z.ZodObject<{
56
80
  bundleDir: z.ZodString;
57
81
  variants: z.ZodEnum<{
@@ -177,6 +201,85 @@ export declare const VerificationReportSchema: z.ZodObject<{
177
201
  ok: z.ZodBoolean;
178
202
  problems: z.ZodArray<z.ZodString>;
179
203
  }, z.core.$strip>;
204
+ /**
205
+ * `diff <bundleA> <bundleB>`'s result. Two `DISPLAY_PATH`s rather than `verify`'s and
206
+ * `inspect`'s single `bundleDir`, because this is the one command that compares two bundles
207
+ * instead of reading one.
208
+ *
209
+ * **`ok` is not tied to whether drift was found.** `verify`'s `ok` means "the hashes
210
+ * matched" because a mismatch there is always a fault — the one thing a bundle promises is
211
+ * that it has not been tampered with. `diff` promises no such thing: it exists precisely
212
+ * because two recordings of the same scenario are *expected* to differ sometimes (an
213
+ * intentional code change) and expected to agree other times (a re-recording taken to prove
214
+ * nothing moved), and neither outcome is a failure of the comparison itself. So `ok` here
215
+ * means "the comparison completed" — both bundles were readable and their event logs
216
+ * parsed — exactly the stance `doctor` takes on a missing voice model ("a state, not a
217
+ * fault"). `identical` is the explicit boolean a caller who wants a verdict reads instead:
218
+ * a CI script gating on drift checks `identical`, not `ok`. `problems` is reserved for what
219
+ * `ok: false` actually means here — a bundle that could not be diffed at all, which reuses
220
+ * `verify`'s exit code 6 (`apps/cli/src/args.ts`'s USAGE table) since it is the same failure
221
+ * class: bundle artifacts that could not be read back out.
222
+ */
223
+ export declare const DiffCommandResultSchema: z.ZodObject<{
224
+ bundleA: z.ZodString;
225
+ bundleB: z.ZodString;
226
+ identical: z.ZodBoolean;
227
+ differences: z.ZodArray<z.ZodObject<{
228
+ category: z.ZodEnum<{
229
+ assertion: "assertion";
230
+ target: "target";
231
+ step: "step";
232
+ timing: "timing";
233
+ caption: "caption";
234
+ }>;
235
+ detail: z.ZodString;
236
+ }, z.core.$strip>>;
237
+ schema: z.ZodLiteral<"agent-demo.result/v1">;
238
+ kind: z.ZodLiteral<"diff">;
239
+ ok: z.ZodBoolean;
240
+ problems: z.ZodArray<z.ZodString>;
241
+ }, z.core.$strip>;
242
+ /**
243
+ * `prune`'s result. Selection never touches disk by itself — `dryRun: true` (the default,
244
+ * absent `--yes`) means `candidates` names what *would* be removed and every one of them is
245
+ * still on disk. `dryRun: false` means `candidates` names what actually *was* removed —
246
+ * **best-effort, not atomic**: `--yes` attempts every selected candidate and does not stop
247
+ * at the first failure, so one bundle a filesystem refuses to delete (permissions, a busy
248
+ * handle) never blocks cleanup of the rest. `candidates` and `bytesReclaimed` report only
249
+ * what is actually gone; a selected candidate `rm` could not remove lands in `failed`
250
+ * instead, with its own path still on disk and the reason why. `ok` is false and the exit
251
+ * code non-zero whenever `failed` is non-empty, so a caller can never mistake a partial run
252
+ * for either a clean one or an empty one — the state is always visible, never silently
253
+ * dropped or misreported as "nothing happened".
254
+ *
255
+ * `skipped` is deliberately separate from both `failed` and `problems` — a directory that
256
+ * `deletionProblem` refuses, or whose manifest cannot be read, was never attempted at all;
257
+ * it is `prune` correctly declining to guess, not a failed deletion. `problems` stays
258
+ * reserved for what makes the *whole* command fail (the same envelope invariant every other
259
+ * result carries): an unreadable workspace root, or — as of `failed` existing — a summary of
260
+ * how many selected candidates a confirmed run could not actually delete.
261
+ */
262
+ export declare const PruneCommandResultSchema: z.ZodObject<{
263
+ dryRun: z.ZodBoolean;
264
+ candidates: z.ZodArray<z.ZodObject<{
265
+ path: z.ZodString;
266
+ bytes: z.ZodNumber;
267
+ scenarioId: z.ZodOptional<z.ZodString>;
268
+ }, z.core.$strip>>;
269
+ skipped: z.ZodArray<z.ZodObject<{
270
+ path: z.ZodString;
271
+ reason: z.ZodString;
272
+ }, z.core.$strip>>;
273
+ failed: z.ZodArray<z.ZodObject<{
274
+ path: z.ZodString;
275
+ reason: z.ZodString;
276
+ }, z.core.$strip>>;
277
+ bytesReclaimed: z.ZodNumber;
278
+ schema: z.ZodLiteral<"agent-demo.result/v1">;
279
+ kind: z.ZodLiteral<"prune">;
280
+ ok: z.ZodBoolean;
281
+ problems: z.ZodArray<z.ZodString>;
282
+ }, z.core.$strip>;
180
283
  export declare const InspectResultSchema: z.ZodObject<{
181
284
  bundleDir: z.ZodString;
182
285
  status: z.ZodEnum<{
@@ -304,13 +407,16 @@ export declare const LicenceResultSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{
304
407
  export type ArtifactRef = z.infer<typeof ArtifactRefSchema>;
305
408
  export type ValidationResult = z.infer<typeof ValidationResultSchema>;
306
409
  export type RunCommandResult = z.infer<typeof RunCommandResultSchema>;
410
+ export type CheckCommandResult = z.infer<typeof CheckCommandResultSchema>;
307
411
  export type RenderCommandResult = z.infer<typeof RenderCommandResultSchema>;
308
412
  export type VerificationReport = z.infer<typeof VerificationReportSchema>;
413
+ export type DiffCommandResult = z.infer<typeof DiffCommandResultSchema>;
414
+ export type PruneCommandResult = z.infer<typeof PruneCommandResultSchema>;
309
415
  export type InspectResult = z.infer<typeof InspectResultSchema>;
310
416
  export type DoctorResult = z.infer<typeof DoctorResultSchema>;
311
417
  export type ActivateResult = z.infer<typeof ActivateResultSchema>;
312
418
  export type LicenceResult = z.infer<typeof LicenceResultSchema>;
313
- export type DemoResult = ValidationResult | RunCommandResult | RenderCommandResult | VerificationReport | InspectResult | DoctorResult | ActivateResult | LicenceResult;
419
+ export type DemoResult = ValidationResult | RunCommandResult | CheckCommandResult | RenderCommandResult | VerificationReport | DiffCommandResult | PruneCommandResult | InspectResult | DoctorResult | ActivateResult | LicenceResult;
314
420
  /**
315
421
  * Renders a filesystem location for display in a result, relative to `root` when it
316
422
  * lies inside it.
@@ -51,6 +51,34 @@ export declare const ScenarioCameraSchema: z.ZodObject<{
51
51
  minDwellMs: z.ZodOptional<z.ZodNumber>;
52
52
  }, z.core.$strip>;
53
53
  export type ScenarioCamera = z.infer<typeof ScenarioCameraSchema>;
54
+ /**
55
+ * Scenario-level narration speed — the one dial this DSL exposes onto the voice, and
56
+ * deliberately the only one. No pitch, no per-step override, no SSML: see
57
+ * `docs/release-checklist.md`'s "no speed, no pitch, no per-step voice, no per-step override,
58
+ * and no SSML" note, of which this closes exactly the first item and nothing else. A per-step
59
+ * override would mean preloading more than one voice at once; `engine.ts` measured a single
60
+ * voice at ~500 MB resident, rising to ~750 MB with a second loaded alongside it — roughly
61
+ * +250 MB per *additional* voice, not 500 MB multiplied by the voice count — a real
62
+ * architecture change this is deliberately not taking on.
63
+ *
64
+ * Nested under `speech` rather than a bare top-level `speed` field: it names what is being
65
+ * tuned (the narration) apart from the DSL's other top-level concerns, matching how `camera`
66
+ * groups framing rather than adding `maxZoom` directly to the metadata object, and it leaves
67
+ * room to grow if narration ever gains a second scenario-level knob without a second unrelated
68
+ * top-level field appearing beside it.
69
+ *
70
+ * `speed` has no measured bound anywhere in this codebase — grepped before picking one: only a
71
+ * hardcoded `speed: 1` default and one `speed: 1.1` in a cache test existed before this field
72
+ * did. Unlike `camera.maxZoom`'s 1.6 (measured against a real page that lost its left third),
73
+ * there is no such measurement to cite for narration speed, so 0.5–2.0 is a clearly-labelled,
74
+ * conservative guess at the range most TTS engines expose a multiplier over, not a fact. A
75
+ * scenario that genuinely needs something outside it is real signal to revisit this bound with
76
+ * a measurement, the way `maxZoom`'s eventually got one.
77
+ */
78
+ export declare const ScenarioSpeechSchema: z.ZodObject<{
79
+ speed: z.ZodOptional<z.ZodNumber>;
80
+ }, z.core.$strip>;
81
+ export type ScenarioSpeech = z.infer<typeof ScenarioSpeechSchema>;
54
82
  export declare const ScenarioMetaSchema: z.ZodObject<{
55
83
  schema: z.ZodLiteral<"agent-demo.scenario/v1">;
56
84
  id: z.ZodString;
@@ -83,6 +111,10 @@ export declare const ScenarioMetaSchema: z.ZodObject<{
83
111
  easeMs: z.ZodOptional<z.ZodNumber>;
84
112
  minDwellMs: z.ZodOptional<z.ZodNumber>;
85
113
  }, z.core.$strip>>;
114
+ speech: z.ZodOptional<z.ZodObject<{
115
+ speed: z.ZodOptional<z.ZodNumber>;
116
+ }, z.core.$strip>>;
117
+ pronunciations: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
86
118
  }, z.core.$strip>;
87
119
  /** Which phase, if any, hands the browser over. Read by the recorder and the surfaces. */
88
120
  export type HandoffMode = z.infer<typeof ScenarioMetaSchema>['handoff'];
package/dist/types.d.ts CHANGED
@@ -15,6 +15,23 @@ export type DemoStep = {
15
15
  target?: Locator;
16
16
  /** Declared interaction kind, recorded on step.start and consumed by the cursor track. */
17
17
  action?: DemoAction;
18
+ /**
19
+ * Dims everything except this step's target rect, with an optional callout label. Recorded
20
+ * on step.start alongside `target`/`action` and consumed by the render-time highlight
21
+ * track — never performed or drawn during capture, which would break byte-reproducibility.
22
+ * Free on every tier, unlike the camera. `true` draws the spotlight alone; `{ label }` adds
23
+ * a callout beside it. Absent means off, which is what every scenario written before this
24
+ * field existed still renders — the same precedent `action` and `holdMs` set.
25
+ *
26
+ * Requires `target` to be measurable: a step that asks for a highlight but whose target's
27
+ * bounding box could not be measured is reported as a diagnostic rather than silently
28
+ * skipped, because unlike a cursor or camera point (which draws on any step, targeted or
29
+ * not, and simply has nothing to draw without a rect) this is an explicit request that went
30
+ * unfulfilled.
31
+ */
32
+ highlight?: boolean | {
33
+ label?: string;
34
+ };
18
35
  /** Extra presentation hold after the action completes, in whole milliseconds. */
19
36
  holdMs?: number;
20
37
  run: () => Promise<unknown>;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@plaintake/scenario",
3
- "version": "1.3.0",
3
+ "version": "1.5.0",
4
4
  "description": "Authoring SDK for PlainTake demo scenarios: defineDemo and the scenario DSL types. Install for editor autocomplete; the PlainTake binary ships a runtime fallback.",
5
5
  "license": "MIT",
6
6
  "type": "module",