@motionscript/audio 0.0.0-stage → 0.1.0-alpha.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +5 -0
- package/LICENSE +201 -0
- package/dist/browser/chunks/chunk-U5OUQJHQ.js +2 -0
- package/dist/browser/chunks/chunk-U5OUQJHQ.js.map +7 -0
- package/dist/browser/index.js +2 -0
- package/dist/browser/index.js.map +7 -0
- package/dist/browser/kit.js +2 -0
- package/dist/browser/kit.js.map +7 -0
- package/dist/browser/manifest.json +12 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +3 -0
- package/dist/index.js.map +1 -0
- package/dist/kit.d.ts +13 -0
- package/dist/kit.d.ts.map +1 -0
- package/dist/kit.js +11 -0
- package/dist/kit.js.map +1 -0
- package/dist/nodes.d.ts +19 -0
- package/dist/nodes.d.ts.map +1 -0
- package/dist/nodes.js +19 -0
- package/dist/nodes.js.map +1 -0
- package/dist/waveform/envelope.d.ts +87 -0
- package/dist/waveform/envelope.d.ts.map +1 -0
- package/dist/waveform/envelope.js +96 -0
- package/dist/waveform/envelope.js.map +1 -0
- package/dist/waveform/full-waveform.d.ts +104 -0
- package/dist/waveform/full-waveform.d.ts.map +1 -0
- package/dist/waveform/full-waveform.js +193 -0
- package/dist/waveform/full-waveform.js.map +1 -0
- package/dist/waveform/index.d.ts +23 -0
- package/dist/waveform/index.d.ts.map +1 -0
- package/dist/waveform/index.js +19 -0
- package/dist/waveform/index.js.map +1 -0
- package/dist/waveform/live-waveform.d.ts +99 -0
- package/dist/waveform/live-waveform.d.ts.map +1 -0
- package/dist/waveform/live-waveform.js +191 -0
- package/dist/waveform/live-waveform.js.map +1 -0
- package/dist/waveform/shared.d.ts +96 -0
- package/dist/waveform/shared.d.ts.map +1 -0
- package/dist/waveform/shared.js +150 -0
- package/dist/waveform/shared.js.map +1 -0
- package/package.json +69 -3
- package/registry.json +28 -0
- package/src/index.ts +2 -0
- package/src/kit.ts +22 -0
- package/src/nodes.ts +19 -0
- package/src/waveform/envelope.ts +143 -0
- package/src/waveform/full-waveform.ts +218 -0
- package/src/waveform/index.ts +30 -0
- package/src/waveform/live-waveform.ts +221 -0
- package/src/waveform/shared.ts +194 -0
- package/README.md +0 -4
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The shape of a recording, as the waveform nodes draw it.
|
|
3
|
+
*
|
|
4
|
+
* ## Why the data is baked rather than listened to
|
|
5
|
+
*
|
|
6
|
+
* A scene has to render identically in the preview and in a headless export,
|
|
7
|
+
* and an export has no audio device, no `AudioContext` and no playback to
|
|
8
|
+
* analyse. So a "live" visualiser cannot read a Web Audio analyser: what it
|
|
9
|
+
* reads is a *pre-computed envelope*, indexed by the scene clock. The picture is
|
|
10
|
+
* then a pure function of time, which is what makes scrubbing backwards, jumping
|
|
11
|
+
* to frame 900, and rendering out of order all produce the same frame.
|
|
12
|
+
*
|
|
13
|
+
* The decode happens in the app, once per file, and arrives here through the
|
|
14
|
+
* build (see `SceneAsset.envelope`) — the same arrangement a chart's rows and a
|
|
15
|
+
* protein's atoms already use, and for the same reason: the build is
|
|
16
|
+
* synchronous and cannot await a fetch.
|
|
17
|
+
*
|
|
18
|
+
* ## Why magnitude rather than min/max
|
|
19
|
+
*
|
|
20
|
+
* An editor's waveform is drawn from per-bucket minima *and* maxima, because an
|
|
21
|
+
* editor is a measuring instrument. These nodes are neither — they are a bar
|
|
22
|
+
* chart of loudness that happens to be shaped like a recording — and audio is
|
|
23
|
+
* near enough symmetric that a mirrored magnitude is indistinguishable at the
|
|
24
|
+
* size a node draws. Half the numbers, and one array to reason about.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* A decoded recording, reduced to a uniform grid of loudness — plus, when the
|
|
29
|
+
* app knows it, who is talking in each bucket.
|
|
30
|
+
*
|
|
31
|
+
* The speaker track is **not** something a decoder can produce; it comes from a
|
|
32
|
+
* diarized transcript and is folded in by the app before the build. It lives
|
|
33
|
+
* here rather than in a parallel structure because every reader of it is also a
|
|
34
|
+
* reader of the magnitudes at the same index, and two arrays that must stay the
|
|
35
|
+
* same length are better as one object than as two arguments.
|
|
36
|
+
*/
|
|
37
|
+
export interface AudioEnvelope {
|
|
38
|
+
/** Peak magnitude per bucket, 0…1. Buckets span {@link duration} evenly. */
|
|
39
|
+
magnitude: Float32Array
|
|
40
|
+
/** The decoded length in seconds. */
|
|
41
|
+
duration: number
|
|
42
|
+
/**
|
|
43
|
+
* Which voice is speaking in each bucket, as an index into the recording's
|
|
44
|
+
* own numbering, or `-1` for "nobody identified". Parallel to
|
|
45
|
+
* {@link magnitude}, or `null` when the track was never diarized.
|
|
46
|
+
*/
|
|
47
|
+
speaker: Int16Array | null
|
|
48
|
+
/** How many distinct voices the diarizer found. Zero when it did not run. */
|
|
49
|
+
speakers: number
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** The envelope of a track that hasn't loaded — or isn't there. */
|
|
53
|
+
export const EMPTY_ENVELOPE: AudioEnvelope = {
|
|
54
|
+
magnitude: new Float32Array(0),
|
|
55
|
+
duration: 0,
|
|
56
|
+
speaker: null,
|
|
57
|
+
speakers: 0,
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** One drawn bar: how loud, and whose. */
|
|
61
|
+
export interface EnvelopeBar {
|
|
62
|
+
/** 0…1. */
|
|
63
|
+
magnitude: number
|
|
64
|
+
/** The voice index, or -1. */
|
|
65
|
+
speaker: number
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Resamples `[from, to)` seconds of an envelope down to `bars` columns.
|
|
70
|
+
*
|
|
71
|
+
* Peak-preserving rather than averaging, which is the whole difference between
|
|
72
|
+
* a waveform and a blur: averaging a hundred buckets into one column flattens
|
|
73
|
+
* every transient, and transients are what makes a drawn recording read as
|
|
74
|
+
* speech rather than as noise. The speaker of a column is the speaker of its
|
|
75
|
+
* *loudest* bucket, on the same reasoning — a column straddling a turn belongs
|
|
76
|
+
* to whoever is actually audible in it.
|
|
77
|
+
*
|
|
78
|
+
* A window past either end of the recording yields silent, unattributed
|
|
79
|
+
* columns rather than clamping onto the nearest real one. That matters for the
|
|
80
|
+
* live visualiser, which is handed a window centred on the playhead and would
|
|
81
|
+
* otherwise smear the first bucket across the whole run-up to a track's start.
|
|
82
|
+
*/
|
|
83
|
+
export function sampleEnvelope(
|
|
84
|
+
envelope: AudioEnvelope,
|
|
85
|
+
from: number,
|
|
86
|
+
to: number,
|
|
87
|
+
bars: number
|
|
88
|
+
): EnvelopeBar[] {
|
|
89
|
+
const count = Math.max(0, Math.floor(bars))
|
|
90
|
+
const out: EnvelopeBar[] = new Array(count)
|
|
91
|
+
for (let i = 0; i < count; i++) out[i] = { magnitude: 0, speaker: -1 }
|
|
92
|
+
|
|
93
|
+
const buckets = envelope.magnitude.length
|
|
94
|
+
if (count === 0 || buckets === 0 || envelope.duration <= 0) return out
|
|
95
|
+
|
|
96
|
+
const perSecond = buckets / envelope.duration
|
|
97
|
+
const span = (to - from) / count
|
|
98
|
+
|
|
99
|
+
for (let i = 0; i < count; i++) {
|
|
100
|
+
const start = (from + span * i) * perSecond
|
|
101
|
+
const end = (from + span * (i + 1)) * perSecond
|
|
102
|
+
|
|
103
|
+
// At least one bucket per column, so a window zoomed in past the grid draws
|
|
104
|
+
// the sample it is standing on rather than an empty band.
|
|
105
|
+
const first = Math.floor(start)
|
|
106
|
+
const last = Math.max(first, Math.ceil(end) - 1)
|
|
107
|
+
|
|
108
|
+
let loudest = 0
|
|
109
|
+
let speaker = -1
|
|
110
|
+
for (let b = first; b <= last; b++) {
|
|
111
|
+
if (b < 0 || b >= buckets) continue
|
|
112
|
+
const value = envelope.magnitude[b]
|
|
113
|
+
if (value < loudest) continue
|
|
114
|
+
loudest = value
|
|
115
|
+
speaker = envelope.speaker ? envelope.speaker[b] : -1
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
out[i] = { magnitude: loudest, speaker }
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
return out
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* The same window, with everything that isn't `voice` silenced.
|
|
126
|
+
*
|
|
127
|
+
* The point of the speaker filter, and the reason it silences rather than
|
|
128
|
+
* *drops*: a podcast visualiser filtered to one host should go quiet while the
|
|
129
|
+
* other one talks, not compress their turn out of existence. Keeping the bars
|
|
130
|
+
* and zeroing them is what makes the picture read as "this person is not
|
|
131
|
+
* speaking right now" instead of as a jump cut.
|
|
132
|
+
*
|
|
133
|
+
* `voice` of `null` returns the bars untouched, which is the unfiltered node.
|
|
134
|
+
*/
|
|
135
|
+
export function forSpeaker(
|
|
136
|
+
bars: EnvelopeBar[],
|
|
137
|
+
voice: number | null
|
|
138
|
+
): EnvelopeBar[] {
|
|
139
|
+
if (voice === null) return bars
|
|
140
|
+
return bars.map((bar) =>
|
|
141
|
+
bar.speaker === voice ? bar : { magnitude: 0, speaker: bar.speaker }
|
|
142
|
+
)
|
|
143
|
+
}
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
import {
|
|
2
|
+
Node2D,
|
|
3
|
+
node,
|
|
4
|
+
command,
|
|
5
|
+
easeInOut,
|
|
6
|
+
fillOps,
|
|
7
|
+
property,
|
|
8
|
+
type AssetScope,
|
|
9
|
+
type CommandArgs,
|
|
10
|
+
type Fill,
|
|
11
|
+
type FillResolved,
|
|
12
|
+
type Node2DProps,
|
|
13
|
+
type RenderContext2D,
|
|
14
|
+
} from "@motionscript/core"
|
|
15
|
+
|
|
16
|
+
import { declarePaints } from "@motionscript/core/component"
|
|
17
|
+
import type { Seekable } from "@motionscript/core/component"
|
|
18
|
+
import { EMPTY_ENVELOPE, sampleEnvelope, type AudioEnvelope } from "./envelope"
|
|
19
|
+
import {
|
|
20
|
+
clamp01,
|
|
21
|
+
groupByPaint,
|
|
22
|
+
layOutBars,
|
|
23
|
+
paintBars,
|
|
24
|
+
voiceColor,
|
|
25
|
+
type WaveformAlignment,
|
|
26
|
+
} from "./shared"
|
|
27
|
+
|
|
28
|
+
export interface FullWaveformProps extends Node2DProps {
|
|
29
|
+
/** The baked recording. See {@link AudioEnvelope} for why it is baked. */
|
|
30
|
+
envelope: AudioEnvelope
|
|
31
|
+
/** How many columns the whole file is drawn as. */
|
|
32
|
+
bars: number
|
|
33
|
+
/** Space between bars, as a fraction of the pitch. See {@link layOutBars}. */
|
|
34
|
+
gap: number
|
|
35
|
+
/** Corner radius of a bar, capped at half its width. */
|
|
36
|
+
radius: number
|
|
37
|
+
/** How much of the file reads as played, `0`–`1`. Tweenable. */
|
|
38
|
+
progress: number
|
|
39
|
+
/** Drive {@link progress} from the node's own clock instead of the prop. */
|
|
40
|
+
follow: boolean
|
|
41
|
+
/** Multiplies every magnitude — a quiet track pulled up to fill its box. */
|
|
42
|
+
gain: number
|
|
43
|
+
/** How much of the box a silent bar still occupies, `0`–`1`. */
|
|
44
|
+
floor: number
|
|
45
|
+
align: WaveformAlignment
|
|
46
|
+
/** Colour each voice's turns rather than splitting on played / not played. */
|
|
47
|
+
byVoice: boolean
|
|
48
|
+
/** A colour per voice index. Short or empty falls back to the palette. */
|
|
49
|
+
voiceColors: string[]
|
|
50
|
+
/** The bars ahead of the playhead. */
|
|
51
|
+
fill: Fill
|
|
52
|
+
/** …and the bars behind it. */
|
|
53
|
+
playedFill: Fill
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* A **waveform overview**: the whole recording's shape, drawn at once.
|
|
58
|
+
*
|
|
59
|
+
* The picture every audio editor opens with, and the one a podcast clip puts
|
|
60
|
+
* under its caption — a static silhouette of the file, with a played portion
|
|
61
|
+
* that advances. It answers "how long is this and where are the loud parts",
|
|
62
|
+
* which is a question about the *file*, not about the moment; its sibling
|
|
63
|
+
* {@link LiveWaveform} answers the other one.
|
|
64
|
+
*
|
|
65
|
+
* ## Progress is authored, not assumed
|
|
66
|
+
*
|
|
67
|
+
* {@link follow} is off by default, and that default is the considered one. A
|
|
68
|
+
* scene has no idea where the recording it is drawing sits on the video's audio
|
|
69
|
+
* lane — a clip is placed, trimmed and sped up out there, and none of that is
|
|
70
|
+
* visible from in here. So the honest default is that the fill is *animation*:
|
|
71
|
+
* the author sweeps it with a command, the same as any other tween, and it lands
|
|
72
|
+
* wherever they put it. `follow` is the shortcut for the common arrangement
|
|
73
|
+
* where the node appears exactly as the track starts, and it says so by being a
|
|
74
|
+
* switch someone has to turn on rather than a behaviour that quietly disagrees
|
|
75
|
+
* with the mix.
|
|
76
|
+
*
|
|
77
|
+
* ## Voices instead of progress
|
|
78
|
+
*
|
|
79
|
+
* With {@link byVoice} on, the played/ahead split is replaced by one colour per
|
|
80
|
+
* speaking voice — the drawing the timeline's clip bars already make, at a size
|
|
81
|
+
* you can read. For a two-hander it is a map of the conversation: who has the
|
|
82
|
+
* floor, where the turns are, how long each one ran. It replaces the progress
|
|
83
|
+
* colours rather than combining with them, because a bar cannot mean two things
|
|
84
|
+
* at once and "played, and also Grace" is a legend nobody can hold in their
|
|
85
|
+
* head.
|
|
86
|
+
*/
|
|
87
|
+
@node({
|
|
88
|
+
key: "fullWaveform",
|
|
89
|
+
parentKey: "node",
|
|
90
|
+
forkable: true,
|
|
91
|
+
layout: {
|
|
92
|
+
children: "freeform",
|
|
93
|
+
defaultWidthMode: "fixed",
|
|
94
|
+
defaultHeightMode: "fixed",
|
|
95
|
+
acceptsChildren: true,
|
|
96
|
+
},
|
|
97
|
+
seed: {
|
|
98
|
+
width: 900,
|
|
99
|
+
height: 160,
|
|
100
|
+
},
|
|
101
|
+
})
|
|
102
|
+
export class FullWaveform extends Node2D<FullWaveformProps> {
|
|
103
|
+
@property({ default: EMPTY_ENVELOPE }) declare envelope: AudioEnvelope
|
|
104
|
+
@property({ default: 120 }) declare bars: number
|
|
105
|
+
@property({ default: 0.3 }) declare gap: number
|
|
106
|
+
@property({ default: 2 }) declare radius: number
|
|
107
|
+
@property({ default: 0 }) declare progress: number
|
|
108
|
+
@property({ default: false }) declare follow: boolean
|
|
109
|
+
@property({ default: 1 }) declare gain: number
|
|
110
|
+
@property({ default: 0.015 }) declare floor: number
|
|
111
|
+
@property({ default: "center" }) declare align: WaveformAlignment
|
|
112
|
+
@property({ default: false }) declare byVoice: boolean
|
|
113
|
+
@property({ default: [] }) declare voiceColors: string[]
|
|
114
|
+
|
|
115
|
+
@property({
|
|
116
|
+
default: "#5b5b7a",
|
|
117
|
+
mapper: fillOps.resolve,
|
|
118
|
+
tween: fillOps.lerp,
|
|
119
|
+
})
|
|
120
|
+
declare fill: Fill
|
|
121
|
+
|
|
122
|
+
@property({
|
|
123
|
+
default: "#6366f1",
|
|
124
|
+
mapper: fillOps.resolve,
|
|
125
|
+
tween: fillOps.lerp,
|
|
126
|
+
})
|
|
127
|
+
declare playedFill: Fill
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* Fill the waveform in, left to right — the node's one command, since
|
|
131
|
+
* everything else about it is expressible as an ordinary `to`.
|
|
132
|
+
*/
|
|
133
|
+
@command()
|
|
134
|
+
sweep(args: CommandArgs<Record<string, never>> & { duration: number }): Seekable {
|
|
135
|
+
return this.to({
|
|
136
|
+
data: { progress: 1 } as Partial<FullWaveformProps>,
|
|
137
|
+
duration: args.duration,
|
|
138
|
+
easing: args.easing ?? easeInOut(),
|
|
139
|
+
})
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/**
|
|
143
|
+
* This node extends `Node2D` rather than `ShapeNode`, so nothing declares the
|
|
144
|
+
* paint props it carries — see {@link declarePaints}. `playedFill` is named
|
|
145
|
+
* past the conventional slots, so it is passed explicitly.
|
|
146
|
+
*/
|
|
147
|
+
override declareAssets(assets: AssetScope): void {
|
|
148
|
+
super.declareAssets(assets)
|
|
149
|
+
declarePaints(this, assets, { fills: [this.playedFill as FillResolved[]] })
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/** Where the fill has reached, from whichever source is in charge. */
|
|
153
|
+
private get played(): number {
|
|
154
|
+
if (!this.follow) return clamp01(this.progress)
|
|
155
|
+
const length = this.envelope.duration
|
|
156
|
+
if (length <= 0) return 0
|
|
157
|
+
return clamp01(this.time.elapsed / length)
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
protected renderSelf(ctx: RenderContext2D): void {
|
|
161
|
+
const rect = this.layoutBounds
|
|
162
|
+
const width = rect?.width ?? 0
|
|
163
|
+
const height = rect?.height ?? 0
|
|
164
|
+
if (width <= 0 || height <= 0) return
|
|
165
|
+
|
|
166
|
+
const count = Math.max(1, Math.round(this.bars))
|
|
167
|
+
const sampled = sampleEnvelope(
|
|
168
|
+
this.envelope,
|
|
169
|
+
0,
|
|
170
|
+
this.envelope.duration,
|
|
171
|
+
count
|
|
172
|
+
)
|
|
173
|
+
const boxes = layOutBars(sampled, {
|
|
174
|
+
width,
|
|
175
|
+
height,
|
|
176
|
+
gap: this.gap,
|
|
177
|
+
align: this.align,
|
|
178
|
+
gain: this.gain,
|
|
179
|
+
floor: this.floor,
|
|
180
|
+
})
|
|
181
|
+
if (boxes.length === 0) return
|
|
182
|
+
|
|
183
|
+
// One draw per colour rather than per bar — see {@link paintBars}. The key
|
|
184
|
+
// is a string because it has to be comparable; the paint it stands for is
|
|
185
|
+
// resolved after the grouping.
|
|
186
|
+
const edge = this.played * boxes.length
|
|
187
|
+
const groups = groupByPaint(boxes, (i) =>
|
|
188
|
+
this.byVoice
|
|
189
|
+
? `voice:${sampled[i].speaker}`
|
|
190
|
+
: i < edge
|
|
191
|
+
? "played"
|
|
192
|
+
: "ahead"
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
for (const [key, run] of groups) {
|
|
196
|
+
const graphics = paintBars(run, this.paintFor(key), this.radius)
|
|
197
|
+
if (graphics) ctx.draw(graphics)
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* The paint a group takes.
|
|
203
|
+
*
|
|
204
|
+
* A voice's colour is a **colour string**, never one of the node's two fill
|
|
205
|
+
* props, and that is deliberate: a voice is identified by hue, so a gradient
|
|
206
|
+
* or an image there would defeat the only thing the colour is for. It also
|
|
207
|
+
* means these paints declare no assets, which is why {@link declareAssets}
|
|
208
|
+
* only has to account for the two real fill props.
|
|
209
|
+
*/
|
|
210
|
+
private paintFor(key: string): Fill {
|
|
211
|
+
if (key === "played") return this.playedFill
|
|
212
|
+
if (key === "ahead") return this.fill
|
|
213
|
+
|
|
214
|
+
const voice = Number(key.slice("voice:".length))
|
|
215
|
+
if (!Number.isFinite(voice) || voice < 0) return this.fill
|
|
216
|
+
return this.voiceColors[voice] ?? voiceColor(voice)
|
|
217
|
+
}
|
|
218
|
+
}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The waveform family: one subject, two questions.
|
|
3
|
+
*
|
|
4
|
+
* - {@link FullWaveform} draws the **file** — the whole recording's silhouette,
|
|
5
|
+
* with a played portion that advances. "How long is this, and where are the
|
|
6
|
+
* loud parts."
|
|
7
|
+
* - {@link LiveWaveform} draws the **moment** — bars moving with playback, the
|
|
8
|
+
* visualiser a podcast clip puts beside the artwork. "Is anything happening
|
|
9
|
+
* right now, and whose voice is it."
|
|
10
|
+
*
|
|
11
|
+
* Both read the same baked {@link AudioEnvelope}, which is why they cannot
|
|
12
|
+
* disagree about a recording, and why neither listens to anything — see that
|
|
13
|
+
* module for why a scene that has to export cannot use a live analyser.
|
|
14
|
+
*/
|
|
15
|
+
export { FullWaveform } from "./full-waveform"
|
|
16
|
+
export type { FullWaveformProps } from "./full-waveform"
|
|
17
|
+
export { LiveWaveform } from "./live-waveform"
|
|
18
|
+
export type { LiveWaveformProps } from "./live-waveform"
|
|
19
|
+
export { EMPTY_ENVELOPE, forSpeaker, sampleEnvelope } from "./envelope"
|
|
20
|
+
export type { AudioEnvelope, EnvelopeBar } from "./envelope"
|
|
21
|
+
export {
|
|
22
|
+
VOICE_PALETTE,
|
|
23
|
+
WAVEFORM_ALIGNMENTS,
|
|
24
|
+
WAVEFORM_STYLES,
|
|
25
|
+
groupByPaint,
|
|
26
|
+
layOutBars,
|
|
27
|
+
paintBars,
|
|
28
|
+
voiceColor,
|
|
29
|
+
} from "./shared"
|
|
30
|
+
export type { BarBox, WaveformAlignment, WaveformStyle } from "./shared"
|
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
import {
|
|
2
|
+
Node2D,
|
|
3
|
+
node,
|
|
4
|
+
fillOps,
|
|
5
|
+
property,
|
|
6
|
+
type AssetScope,
|
|
7
|
+
type Fill,
|
|
8
|
+
type Node2DProps,
|
|
9
|
+
type RenderContext2D,
|
|
10
|
+
} from "@motionscript/core"
|
|
11
|
+
|
|
12
|
+
import { declarePaints } from "@motionscript/core/component"
|
|
13
|
+
import {
|
|
14
|
+
EMPTY_ENVELOPE,
|
|
15
|
+
forSpeaker,
|
|
16
|
+
sampleEnvelope,
|
|
17
|
+
type AudioEnvelope,
|
|
18
|
+
type EnvelopeBar,
|
|
19
|
+
} from "./envelope"
|
|
20
|
+
import {
|
|
21
|
+
clamp01,
|
|
22
|
+
groupByPaint,
|
|
23
|
+
layOutBars,
|
|
24
|
+
livePaintKey,
|
|
25
|
+
paintBars,
|
|
26
|
+
voiceColor,
|
|
27
|
+
type WaveformAlignment,
|
|
28
|
+
type WaveformStyle,
|
|
29
|
+
} from "./shared"
|
|
30
|
+
|
|
31
|
+
export interface LiveWaveformProps extends Node2DProps {
|
|
32
|
+
/** The baked recording. See {@link AudioEnvelope} for why it is baked. */
|
|
33
|
+
envelope: AudioEnvelope
|
|
34
|
+
/** How many bars the visualiser is made of. */
|
|
35
|
+
bars: number
|
|
36
|
+
/** Space between bars, as a fraction of the pitch. */
|
|
37
|
+
gap: number
|
|
38
|
+
radius: number
|
|
39
|
+
/** How many seconds of audio are on screen at once. */
|
|
40
|
+
window: number
|
|
41
|
+
/** Seconds into the file the node's first frame shows. */
|
|
42
|
+
offset: number
|
|
43
|
+
/** Advance with the scene clock. Off freezes the picture at {@link offset}. */
|
|
44
|
+
playing: boolean
|
|
45
|
+
/** Multiplies every magnitude before it is clamped. */
|
|
46
|
+
gain: number
|
|
47
|
+
/** How much of the box a silent bar still occupies, `0`–`1`. */
|
|
48
|
+
floor: number
|
|
49
|
+
align: WaveformAlignment
|
|
50
|
+
style: WaveformStyle
|
|
51
|
+
/**
|
|
52
|
+
* Show only this voice's activity, by index into the recording's own
|
|
53
|
+
* numbering. `-1` — the default — is every voice.
|
|
54
|
+
*/
|
|
55
|
+
speaker: number
|
|
56
|
+
/** Colour each bar by whoever is speaking in it. */
|
|
57
|
+
byVoice: boolean
|
|
58
|
+
/** A colour per voice index. Short or empty falls back to the palette. */
|
|
59
|
+
voiceColors: string[]
|
|
60
|
+
fill: Fill
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* A **live audio visualiser**: bars that move with the recording as it plays.
|
|
65
|
+
*
|
|
66
|
+
* The thing every podcast-to-video tool puts beside the artwork. It is not a
|
|
67
|
+
* picture of the file — that is {@link FullWaveform} — it is a picture of *this
|
|
68
|
+
* moment*, and what it is for is making a static frame of a talking head feel
|
|
69
|
+
* like it is playing.
|
|
70
|
+
*
|
|
71
|
+
* ## "Live" without listening
|
|
72
|
+
*
|
|
73
|
+
* Nothing here reads an audio device. The bars are sampled out of a baked
|
|
74
|
+
* envelope at the node's own elapsed time, which is what makes the visualiser
|
|
75
|
+
* survive the thing it exists for: an **export**, where there is no playback to
|
|
76
|
+
* analyse and frames may be rendered out of order or in parallel. A frame is a
|
|
77
|
+
* pure function of its time, so scrubbing back and forth over the visualiser
|
|
78
|
+
* redraws exactly what it drew the first time — which an analyser node, by
|
|
79
|
+
* construction, cannot promise. See {@link AudioEnvelope}.
|
|
80
|
+
*
|
|
81
|
+
* ## The speaker filter
|
|
82
|
+
*
|
|
83
|
+
* {@link speaker} narrows the node to one voice of a diarized recording, and it
|
|
84
|
+
* is the reason this node knows about transcripts at all. Two visualisers beside
|
|
85
|
+
* two headshots, each filtered to its own host, is a podcast edit that would
|
|
86
|
+
* otherwise be done by hand for every turn — and because the filter *silences*
|
|
87
|
+
* rather than skips (see {@link forSpeaker}), the quiet host visibly stops
|
|
88
|
+
* talking instead of vanishing. Nothing about that requires the words: only who
|
|
89
|
+
* is talking when, which is what the diarizer produces.
|
|
90
|
+
*
|
|
91
|
+
* A filter naming a voice the recording doesn't have draws a resting row rather
|
|
92
|
+
* than everything — a picture of "this person never speaks here", which is
|
|
93
|
+
* true, where falling back to the unfiltered mix would be a silent lie about
|
|
94
|
+
* whose voice you were watching.
|
|
95
|
+
*/
|
|
96
|
+
@node({
|
|
97
|
+
key: "liveWaveform",
|
|
98
|
+
parentKey: "node",
|
|
99
|
+
forkable: true,
|
|
100
|
+
layout: {
|
|
101
|
+
children: "freeform",
|
|
102
|
+
defaultWidthMode: "fixed",
|
|
103
|
+
defaultHeightMode: "fixed",
|
|
104
|
+
acceptsChildren: true,
|
|
105
|
+
},
|
|
106
|
+
seed: {
|
|
107
|
+
width: 420,
|
|
108
|
+
height: 220,
|
|
109
|
+
},
|
|
110
|
+
})
|
|
111
|
+
export class LiveWaveform extends Node2D<LiveWaveformProps> {
|
|
112
|
+
@property({ default: EMPTY_ENVELOPE }) declare envelope: AudioEnvelope
|
|
113
|
+
@property({ default: 48 }) declare bars: number
|
|
114
|
+
@property({ default: 0.35 }) declare gap: number
|
|
115
|
+
@property({ default: 4 }) declare radius: number
|
|
116
|
+
@property({ default: 3 }) declare window: number
|
|
117
|
+
@property({ default: 0 }) declare offset: number
|
|
118
|
+
@property({ default: true }) declare playing: boolean
|
|
119
|
+
@property({ default: 1.4 }) declare gain: number
|
|
120
|
+
@property({ default: 0.04 }) declare floor: number
|
|
121
|
+
@property({ default: "center" }) declare align: WaveformAlignment
|
|
122
|
+
@property({ default: "bars" }) declare style: WaveformStyle
|
|
123
|
+
@property({ default: -1 }) declare speaker: number
|
|
124
|
+
@property({ default: false }) declare byVoice: boolean
|
|
125
|
+
@property({ default: [] }) declare voiceColors: string[]
|
|
126
|
+
|
|
127
|
+
@property({
|
|
128
|
+
default: "#6366f1",
|
|
129
|
+
mapper: fillOps.resolve,
|
|
130
|
+
tween: fillOps.lerp,
|
|
131
|
+
})
|
|
132
|
+
declare fill: Fill
|
|
133
|
+
|
|
134
|
+
override declareAssets(assets: AssetScope): void {
|
|
135
|
+
super.declareAssets(assets)
|
|
136
|
+
declarePaints(this, assets)
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* The second of the recording the visualiser is showing.
|
|
141
|
+
*
|
|
142
|
+
* Off the node's own clock rather than the scene's, which is what "from the
|
|
143
|
+
* moment this node appeared" means — the same clock motion-script's own
|
|
144
|
+
* `Video` times its picture from, and the same one that makes a node dropped
|
|
145
|
+
* halfway through a scene start its recording at its entrance rather than
|
|
146
|
+
* partway in.
|
|
147
|
+
*/
|
|
148
|
+
private get at(): number {
|
|
149
|
+
return this.offset + (this.playing ? Math.max(0, this.time.elapsed) : 0)
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
protected renderSelf(ctx: RenderContext2D): void {
|
|
153
|
+
const rect = this.layoutBounds
|
|
154
|
+
const width = rect?.width ?? 0
|
|
155
|
+
const height = rect?.height ?? 0
|
|
156
|
+
if (width <= 0 || height <= 0) return
|
|
157
|
+
|
|
158
|
+
const count = Math.max(1, Math.round(this.bars))
|
|
159
|
+
const span = Math.max(0.05, this.window)
|
|
160
|
+
const at = this.at
|
|
161
|
+
const voice = this.speaker < 0 ? null : Math.round(this.speaker)
|
|
162
|
+
|
|
163
|
+
// The window is centred on the playhead, so the bars run *toward* the
|
|
164
|
+
// middle and away again — the arrangement every mixer's meter uses, and the
|
|
165
|
+
// one that reads as sound arriving rather than as a strip scrolling past.
|
|
166
|
+
const sampled = forSpeaker(
|
|
167
|
+
sampleEnvelope(this.envelope, at - span / 2, at + span / 2, count),
|
|
168
|
+
voice
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
const shaped = this.style === "line" ? taper(sampled) : sampled
|
|
172
|
+
const boxes = layOutBars(shaped, {
|
|
173
|
+
width,
|
|
174
|
+
height,
|
|
175
|
+
gap: this.style === "blocks" ? 0.12 : this.gap,
|
|
176
|
+
align: this.align,
|
|
177
|
+
gain: this.gain,
|
|
178
|
+
floor: this.floor,
|
|
179
|
+
})
|
|
180
|
+
if (boxes.length === 0) return
|
|
181
|
+
|
|
182
|
+
const groups = groupByPaint(boxes, (i) =>
|
|
183
|
+
livePaintKey(shaped[i].speaker, this.byVoice, voice)
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
for (const [key, run] of groups) {
|
|
187
|
+
const graphics = paintBars(run, this.paintFor(key), this.radius)
|
|
188
|
+
if (graphics) ctx.draw(graphics)
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/** See {@link FullWaveform.paintFor} — a voice is a hue, never a gradient. */
|
|
193
|
+
private paintFor(key: string): Fill {
|
|
194
|
+
if (key === "all") return this.fill
|
|
195
|
+
const voice = Number(key.slice("voice:".length))
|
|
196
|
+
if (!Number.isFinite(voice) || voice < 0) return this.fill
|
|
197
|
+
return this.voiceColors[voice] ?? voiceColor(voice)
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* Rolls the ends of a window off toward silence.
|
|
203
|
+
*
|
|
204
|
+
* What separates the `line` style from `bars`: a hard-edged window makes the
|
|
205
|
+
* loud bar entering at the right edge appear from nothing, which reads as a
|
|
206
|
+
* glitch rather than as sound arriving. Fading the outer fifth turns the same
|
|
207
|
+
* data into a shape that grows out of the baseline — the profile a spectrum
|
|
208
|
+
* display has, and the reason it looks like one object rather than a row of
|
|
209
|
+
* separate meters.
|
|
210
|
+
*/
|
|
211
|
+
function taper(bars: readonly EnvelopeBar[]): EnvelopeBar[] {
|
|
212
|
+
const count = bars.length
|
|
213
|
+
if (count < 3) return [...bars]
|
|
214
|
+
const edge = Math.max(1, Math.floor(count / 5))
|
|
215
|
+
|
|
216
|
+
return bars.map((bar, i) => {
|
|
217
|
+
const fromEnd = Math.min(i, count - 1 - i)
|
|
218
|
+
const weight = clamp01(fromEnd / edge)
|
|
219
|
+
return { magnitude: bar.magnitude * weight, speaker: bar.speaker }
|
|
220
|
+
})
|
|
221
|
+
}
|