creator-editing-studio 1.2.0 → 1.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,465 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Best-take selection: turn a script plus a messy take into a clean cut plan.
|
|
4
|
+
*
|
|
5
|
+
* npm run takes -- <script.md|txt> <words.json> [--project <name>] [--out <file>]
|
|
6
|
+
* [--fps 30] [--gap-keep 0.18] [--min-coverage 0.6] [--loose]
|
|
7
|
+
*
|
|
8
|
+
* THE PROBLEM THIS SOLVES. A talking head is read from a script, so the speaker fumbles and
|
|
9
|
+
* says a sentence again. `trim.mjs` removes silence and filler, but it keeps every attempt —
|
|
10
|
+
* so the cut contains the stumble AND the fix, and someone watches the whole thing to find
|
|
11
|
+
* them. On the setup video that meant 16 drop ranges chosen by hand, several missed, and a
|
|
12
|
+
* trip to CapCut to take out repeats nobody caught.
|
|
13
|
+
*
|
|
14
|
+
* WHY NOT JUST HUNT FOR REPEATS. Repeat-hunting is a filter: it finds duplicates it recognises
|
|
15
|
+
* and misses the rest. Anchoring to the script is a constructor: each sentence of the script is
|
|
16
|
+
* an atom, every atom gets exactly one span of audio, and anything that is not the winning span
|
|
17
|
+
* for some atom never enters the cut. Repeats are not removed — they are impossible. Filler,
|
|
18
|
+
* half-sentences, asides and dead air are gone for the same reason, without a rule for each.
|
|
19
|
+
*
|
|
20
|
+
* It also fails loudly. If a script sentence has no usable take, the build stops and names the
|
|
21
|
+
* line, instead of handing over a cut with a hole for someone to find on playback.
|
|
22
|
+
*
|
|
23
|
+
* WHAT IT WRITES work/<project>/<stem>.takes.json
|
|
24
|
+
* keep[] {from, to, atom} source seconds to use, in order
|
|
25
|
+
* dropped[] {from, to, reason, text}
|
|
26
|
+
* atoms[] every script sentence, its chosen take and the attempts that lost
|
|
27
|
+
* checks{} the verification results, all of which must pass
|
|
28
|
+
*
|
|
29
|
+
* The shape of `keep[]` matches `trim.mjs`, so the rest of the pipeline is unchanged.
|
|
30
|
+
*/
|
|
31
|
+
import {existsSync, mkdirSync, readFileSync, writeFileSync} from 'node:fs';
|
|
32
|
+
import {basename, dirname, join} from 'node:path';
|
|
33
|
+
|
|
34
|
+
/* ------------------------------------------------------------------------ args */
|
|
35
|
+
|
|
36
|
+
const args = process.argv.slice(2);
|
|
37
|
+
const flag = (name, fallback) => {
|
|
38
|
+
const i = args.indexOf(`--${name}`);
|
|
39
|
+
return i === -1 ? fallback : args[i + 1];
|
|
40
|
+
};
|
|
41
|
+
const has = (name) => args.includes(`--${name}`);
|
|
42
|
+
const positional = args.filter((a, i) => !a.startsWith('--') && !String(args[i - 1] ?? '').startsWith('--'));
|
|
43
|
+
|
|
44
|
+
const [scriptPath, wordsPath] = positional;
|
|
45
|
+
if (!scriptPath || !wordsPath) {
|
|
46
|
+
console.error('usage: npm run takes -- <script.md|txt> <words.json> [--project <name>]');
|
|
47
|
+
process.exit(1);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const FPS = Number(flag('fps', 30));
|
|
51
|
+
/** A gap longer than this between kept spans is collapsed down to it. */
|
|
52
|
+
const GAP_KEEP = Number(flag('gap-keep', 0.18));
|
|
53
|
+
/** Below this share of an atom's words matched, a span is not a take of that atom at all. */
|
|
54
|
+
const MIN_COVERAGE = Number(flag('min-coverage', 0.6));
|
|
55
|
+
/** Padding kept around a span so cuts do not clip the first and last phoneme. */
|
|
56
|
+
const PAD = 0.06;
|
|
57
|
+
|
|
58
|
+
/* ------------------------------------------------------------- script and words */
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* The script as atoms.
|
|
62
|
+
*
|
|
63
|
+
* A sentence is the right unit: small enough that one fumble does not cost a paragraph, big
|
|
64
|
+
* enough to identify uniquely in the transcript. Markdown furniture and the BRIEF.md comment
|
|
65
|
+
* blocks are stripped — they are instructions to the writer, never spoken.
|
|
66
|
+
*/
|
|
67
|
+
const atomise = (raw) =>
|
|
68
|
+
raw
|
|
69
|
+
// Everything above the first `---` is front matter or a note to whoever wrote the script.
|
|
70
|
+
// It is never spoken, and counting it as missing dialogue buries the lines that really are.
|
|
71
|
+
.split(/^---\s*$/m)
|
|
72
|
+
.slice(1)
|
|
73
|
+
.join('\n')
|
|
74
|
+
.replace(/<!--[\s\S]*?-->/g, '')
|
|
75
|
+
.replace(/^#.*$/gm, '')
|
|
76
|
+
.replace(/^[-*|>].*$/gm, '')
|
|
77
|
+
.replace(/`[^`]*`/g, ' ')
|
|
78
|
+
.split(/(?<=[.?!])\s+/)
|
|
79
|
+
.map((s) => s.replace(/\s+/g, ' ').trim())
|
|
80
|
+
.filter((s) => s.split(/\s+/).filter(Boolean).length >= 3);
|
|
81
|
+
|
|
82
|
+
/** Comparable key for a word. Mirrors align_script.py so both agree on what "same" means. */
|
|
83
|
+
/**
|
|
84
|
+
* Spellings that mean the same word.
|
|
85
|
+
*
|
|
86
|
+
* The script is written in British English and the recogniser answers in American. Without this,
|
|
87
|
+
* "licence" never matches "license" and a correctly spoken sentence is reported as missing while
|
|
88
|
+
* the audio for it is thrown away as off-script. That is a worse failure than a repeat.
|
|
89
|
+
*/
|
|
90
|
+
const SPELLING = new Map(Object.entries({
|
|
91
|
+
license: 'licence', color: 'colour', colors: 'colours', favorite: 'favourite',
|
|
92
|
+
realize: 'realise', organize: 'organise', gray: 'grey', analyze: 'analyse',
|
|
93
|
+
}));
|
|
94
|
+
|
|
95
|
+
const key = (token) => {
|
|
96
|
+
const bare = token.toLowerCase().replace(/['‘’]/g, '').replace(/[^a-z0-9]/g, '');
|
|
97
|
+
return SPELLING.get(bare) ?? bare;
|
|
98
|
+
};
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Contractions, expanded.
|
|
102
|
+
*
|
|
103
|
+
* The script says "it'll" and the speaker says "it will" — or the other way round, depending on
|
|
104
|
+
* the day and the recogniser. Both sides expand to the same pair of words so they compare equal.
|
|
105
|
+
* Only forms carrying an apostrophe are touched, so "well" is never mistaken for "we will".
|
|
106
|
+
*/
|
|
107
|
+
const CONTRACTION = new Map(Object.entries({
|
|
108
|
+
itll: ['it', 'will'], theyll: ['they', 'will'], youll: ['you', 'will'], well: ['we', 'will'],
|
|
109
|
+
itsa: ['it', 'is', 'a'], its: ['it', 'is'], thats: ['that', 'is'], whats: ['what', 'is'],
|
|
110
|
+
theres: ['there', 'is'], heres: ['here', 'is'], lets: ['let', 'us'], youre: ['you', 'are'],
|
|
111
|
+
theyre: ['they', 'are'], weve: ['we', 'have'], youve: ['you', 'have'], ive: ['i', 'have'],
|
|
112
|
+
dont: ['do', 'not'], doesnt: ['does', 'not'], didnt: ['did', 'not'], isnt: ['is', 'not'],
|
|
113
|
+
wasnt: ['was', 'not'], arent: ['are', 'not'], wont: ['will', 'not'], cant: ['can', 'not'],
|
|
114
|
+
couldnt: ['could', 'not'], wouldnt: ['would', 'not'], shouldnt: ['should', 'not'],
|
|
115
|
+
youd: ['you', 'would'], id: ['i', 'would'], wed: ['we', 'would'], ill: ['i', 'will'],
|
|
116
|
+
}));
|
|
117
|
+
|
|
118
|
+
/** One spoken or written word becomes the one or more keys it is equivalent to. */
|
|
119
|
+
const keysOfWord = (token) => {
|
|
120
|
+
const k = key(token);
|
|
121
|
+
if (!k) return [];
|
|
122
|
+
// Only expand when the original actually had an apostrophe, so ordinary words are left alone.
|
|
123
|
+
if (/['‘’]/.test(token) && CONTRACTION.has(k)) return CONTRACTION.get(k);
|
|
124
|
+
return [k];
|
|
125
|
+
};
|
|
126
|
+
|
|
127
|
+
const keysOf = (text) => text.split(/[\s\-/]+/).flatMap(keysOfWord);
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* Filler. Deliberately short.
|
|
131
|
+
*
|
|
132
|
+
* Stripping every "so" and "you know" makes speech sound sanded down, which reads as
|
|
133
|
+
* over-edited and is its own kind of amateur. These are only ever used to PENALISE a take,
|
|
134
|
+
* never to cut inside a winning one — if the best attempt at a line contains an "um", the line
|
|
135
|
+
* keeps its "um" and sounds like a person.
|
|
136
|
+
*/
|
|
137
|
+
const FILLER = new Set(['um', 'uh', 'uhh', 'umm', 'erm', 'hmm', 'mmm', 'ah', 'eh']);
|
|
138
|
+
|
|
139
|
+
const words = JSON.parse(readFileSync(wordsPath, 'utf8')).words.filter((w) => w.start != null && w.end != null);
|
|
140
|
+
const rawScript = readFileSync(scriptPath, 'utf8');
|
|
141
|
+
// A plain .txt script has no `---` rule, so give it one — atomise() always drops what is above
|
|
142
|
+
// the first rule, and without this a bare script would come back empty.
|
|
143
|
+
const atoms = atomise(/^---\s*$/m.test(rawScript) ? rawScript : `---\n${rawScript}`);
|
|
144
|
+
/** Each spoken word as the list of keys it is equivalent to. Index stays 1:1 with `words`. */
|
|
145
|
+
const tokenKeys = words.map((w) => {
|
|
146
|
+
const ks = keysOfWord(w.word);
|
|
147
|
+
return ks.length ? ks : ['\u0000'];
|
|
148
|
+
});
|
|
149
|
+
const tokens = tokenKeys.map((ks) => ks[0]);
|
|
150
|
+
|
|
151
|
+
if (!atoms.length) {
|
|
152
|
+
console.error(`No script sentences found in ${scriptPath}.`);
|
|
153
|
+
process.exit(1);
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/* --------------------------------------------------------------- finding takes */
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Every span of transcript that is a plausible attempt at `atom`.
|
|
160
|
+
*
|
|
161
|
+
* Greedy forward alignment from each position whose word matches the atom's first word, allowing
|
|
162
|
+
* the speaker to skip, swap or insert a few words. A take is scored on how much of the sentence
|
|
163
|
+
* actually made it out, how sure the aligner was, and whether it ran smoothly — a long pause or
|
|
164
|
+
* an "um" inside a sentence is exactly what a fumbled attempt sounds like.
|
|
165
|
+
*/
|
|
166
|
+
const findTakes = (atom, atomIndex) => {
|
|
167
|
+
const want = keysOf(atom);
|
|
168
|
+
if (!want.length) return [];
|
|
169
|
+
const out = [];
|
|
170
|
+
/** How far past the sentence the speaker may wander before the attempt is abandoned. */
|
|
171
|
+
const window = Math.ceil(want.length * 2.2) + 6;
|
|
172
|
+
|
|
173
|
+
/**
|
|
174
|
+
* A take may begin on any of the sentence's first three words.
|
|
175
|
+
*
|
|
176
|
+
* Anchoring only on the first word loses a whole sentence to a single swapped article — the
|
|
177
|
+
* script says "This system gets better", the speaker says "The system gets better", and the
|
|
178
|
+
* line is reported missing while its audio is discarded. Coverage still has to clear the
|
|
179
|
+
* threshold, so a take that drops its opening words scores lower rather than passing free.
|
|
180
|
+
*/
|
|
181
|
+
const anchors = want.slice(0, 3);
|
|
182
|
+
|
|
183
|
+
for (let start = 0; start < tokens.length; start++) {
|
|
184
|
+
const anchorAt = anchors.indexOf(tokens[start]);
|
|
185
|
+
if (anchorAt === -1) continue;
|
|
186
|
+
|
|
187
|
+
let wi = anchorAt;
|
|
188
|
+
let matched = anchorAt;
|
|
189
|
+
let last = start;
|
|
190
|
+
let extras = 0;
|
|
191
|
+
for (let ti = start; ti < Math.min(tokens.length, start + window) && wi < want.length; ti++) {
|
|
192
|
+
// A contraction covers more than one script word, so consume as many as it supplies.
|
|
193
|
+
let consumed = 0;
|
|
194
|
+
while (consumed < tokenKeys[ti].length && wi + consumed < want.length && tokenKeys[ti][consumed] === want[wi + consumed]) consumed++;
|
|
195
|
+
if (consumed > 0) {
|
|
196
|
+
matched += consumed;
|
|
197
|
+
wi += consumed;
|
|
198
|
+
last = ti;
|
|
199
|
+
} else if (want.indexOf(tokens[ti], wi) !== -1 && want.indexOf(tokens[ti], wi) - wi <= 3) {
|
|
200
|
+
// The speaker skipped a word or two and carried on; follow them.
|
|
201
|
+
wi = want.indexOf(tokens[ti], wi) + 1;
|
|
202
|
+
matched++;
|
|
203
|
+
last = ti;
|
|
204
|
+
} else {
|
|
205
|
+
extras++;
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
const coverage = matched / want.length;
|
|
210
|
+
if (coverage < MIN_COVERAGE) continue;
|
|
211
|
+
|
|
212
|
+
const from = words[start].start;
|
|
213
|
+
const to = words[last].end;
|
|
214
|
+
const span = words.slice(start, last + 1);
|
|
215
|
+
|
|
216
|
+
let worstGap = 0;
|
|
217
|
+
for (let i = 1; i < span.length; i++) worstGap = Math.max(worstGap, span[i].start - span[i - 1].end);
|
|
218
|
+
const fillerInside = span.filter((w) => FILLER.has(key(w.word))).length;
|
|
219
|
+
const confidence = span.reduce((a, w) => a + (w.score ?? 0.8), 0) / span.length;
|
|
220
|
+
|
|
221
|
+
out.push({
|
|
222
|
+
atomIndex,
|
|
223
|
+
startIndex: start,
|
|
224
|
+
endIndex: last,
|
|
225
|
+
from,
|
|
226
|
+
to,
|
|
227
|
+
coverage,
|
|
228
|
+
confidence,
|
|
229
|
+
worstGap,
|
|
230
|
+
fillerInside,
|
|
231
|
+
extras,
|
|
232
|
+
text: span.map((w) => w.word).join(' '),
|
|
233
|
+
/**
|
|
234
|
+
* One number, so the chooser has something to maximise.
|
|
235
|
+
*
|
|
236
|
+
* Coverage dominates: a complete sentence beats a smooth fragment every time. The small
|
|
237
|
+
* bonus for starting later is there because when someone fumbles they fix it on the next
|
|
238
|
+
* try, so with two otherwise equal attempts the second is the keeper.
|
|
239
|
+
*/
|
|
240
|
+
score:
|
|
241
|
+
coverage * 10 +
|
|
242
|
+
confidence * 2 -
|
|
243
|
+
Math.max(0, worstGap - 0.4) * 3 -
|
|
244
|
+
fillerInside * 0.8 -
|
|
245
|
+
(extras / want.length) * 1.5 +
|
|
246
|
+
(start / tokens.length) * 0.3,
|
|
247
|
+
});
|
|
248
|
+
|
|
249
|
+
start = last; // attempts cannot overlap each other
|
|
250
|
+
}
|
|
251
|
+
return out;
|
|
252
|
+
};
|
|
253
|
+
|
|
254
|
+
const takesPerAtom = atoms.map((atom, i) => findTakes(atom, i));
|
|
255
|
+
|
|
256
|
+
/* -------------------------------------------------------------- choosing takes */
|
|
257
|
+
|
|
258
|
+
/**
|
|
259
|
+
* One take per atom, in script order, never overlapping.
|
|
260
|
+
*
|
|
261
|
+
* Greedy picking of each atom's best take breaks as soon as the best take of sentence 4 sits
|
|
262
|
+
* earlier in the tape than the chosen take of sentence 5. This is weighted interval scheduling
|
|
263
|
+
* over the atoms in order: for each candidate, the best total reachable while ending before the
|
|
264
|
+
* next atom starts. Small enough to be exact rather than heuristic.
|
|
265
|
+
*/
|
|
266
|
+
const NEG = -1e9;
|
|
267
|
+
const best = takesPerAtom.map((takes) => takes.map(() => ({total: NEG, prev: -1})));
|
|
268
|
+
|
|
269
|
+
for (let a = 0; a < atoms.length; a++) {
|
|
270
|
+
const takes = takesPerAtom[a];
|
|
271
|
+
for (let c = 0; c < takes.length; c++) {
|
|
272
|
+
if (a === 0) {
|
|
273
|
+
best[a][c] = {total: takes[c].score, prev: -1};
|
|
274
|
+
continue;
|
|
275
|
+
}
|
|
276
|
+
let bestPrev = NEG;
|
|
277
|
+
let bestPrevIndex = -1;
|
|
278
|
+
for (let p = 0; p < takesPerAtom[a - 1].length; p++) {
|
|
279
|
+
if (best[a - 1][p].total === NEG) continue;
|
|
280
|
+
if (takesPerAtom[a - 1][p].endIndex >= takes[c].startIndex) continue;
|
|
281
|
+
if (best[a - 1][p].total > bestPrev) {
|
|
282
|
+
bestPrev = best[a - 1][p].total;
|
|
283
|
+
bestPrevIndex = p;
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
// An atom with no reachable predecessor can still start a run; the coverage check below
|
|
287
|
+
// is what decides whether the result is acceptable, not this.
|
|
288
|
+
best[a][c] =
|
|
289
|
+
bestPrevIndex === -1
|
|
290
|
+
? {total: takes[c].score, prev: -1}
|
|
291
|
+
: {total: bestPrev + takes[c].score, prev: bestPrevIndex};
|
|
292
|
+
}
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
const chosen = new Array(atoms.length).fill(null);
|
|
296
|
+
let cursor = -1;
|
|
297
|
+
for (let a = atoms.length - 1; a >= 0; a--) {
|
|
298
|
+
const takes = takesPerAtom[a];
|
|
299
|
+
if (!takes.length) continue;
|
|
300
|
+
let pick = cursor;
|
|
301
|
+
if (pick === -1) {
|
|
302
|
+
let top = NEG;
|
|
303
|
+
for (let c = 0; c < takes.length; c++)
|
|
304
|
+
if (best[a][c].total > top) {
|
|
305
|
+
top = best[a][c].total;
|
|
306
|
+
pick = c;
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
if (pick === -1 || !takes[pick]) continue;
|
|
310
|
+
chosen[a] = takes[pick];
|
|
311
|
+
cursor = best[a][pick].prev;
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
/* ------------------------------------------------------------------ the result */
|
|
315
|
+
|
|
316
|
+
const missing = [];
|
|
317
|
+
atoms.forEach((atom, i) => {
|
|
318
|
+
if (!chosen[i]) missing.push({index: i, text: atom, attempts: takesPerAtom[i].length});
|
|
319
|
+
});
|
|
320
|
+
|
|
321
|
+
const keep = chosen
|
|
322
|
+
.filter(Boolean)
|
|
323
|
+
.map((t) => ({from: Math.max(0, t.from - PAD), to: t.to + PAD, atom: t.atomIndex, text: t.text}))
|
|
324
|
+
.sort((a, b) => a.from - b.from);
|
|
325
|
+
|
|
326
|
+
// Overlapping padding between neighbours would double a word; meet in the middle instead.
|
|
327
|
+
for (let i = 1; i < keep.length; i++) {
|
|
328
|
+
if (keep[i].from < keep[i - 1].to) {
|
|
329
|
+
const mid = (keep[i].from + keep[i - 1].to) / 2;
|
|
330
|
+
keep[i - 1].to = mid;
|
|
331
|
+
keep[i].from = mid;
|
|
332
|
+
}
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
const keptIndexes = new Set();
|
|
336
|
+
chosen.filter(Boolean).forEach((t) => {
|
|
337
|
+
for (let i = t.startIndex; i <= t.endIndex; i++) keptIndexes.add(i);
|
|
338
|
+
});
|
|
339
|
+
|
|
340
|
+
const dropped = [];
|
|
341
|
+
let runStart = null;
|
|
342
|
+
for (let i = 0; i < words.length; i++) {
|
|
343
|
+
const inCut = keptIndexes.has(i);
|
|
344
|
+
if (!inCut && runStart === null) runStart = i;
|
|
345
|
+
if ((inCut || i === words.length - 1) && runStart !== null) {
|
|
346
|
+
const end = inCut ? i - 1 : i;
|
|
347
|
+
const text = words.slice(runStart, end + 1).map((w) => w.word).join(' ');
|
|
348
|
+
const alsoKept = chosen.filter(Boolean).some((t) => keysOf(t.text).join(' ').includes(keysOf(text).join(' ')));
|
|
349
|
+
dropped.push({
|
|
350
|
+
from: words[runStart].start,
|
|
351
|
+
to: words[end].end,
|
|
352
|
+
reason: alsoKept ? 'retake' : text.split(/\s+/).length <= 2 ? 'filler' : 'off-script',
|
|
353
|
+
text,
|
|
354
|
+
});
|
|
355
|
+
runStart = null;
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
/* ----------------------------------------------------------------- verification
|
|
360
|
+
* Run on the transcript the cut WILL have, worked out from the spans kept, so there is no
|
|
361
|
+
* second transcription and no waiting. Every one of these must pass; a build that fails here
|
|
362
|
+
* is a build that would have sent someone to CapCut. */
|
|
363
|
+
|
|
364
|
+
const outWords = [...keptIndexes].sort((a, b) => a - b).map((i) => words[i]);
|
|
365
|
+
const outKeys = outWords.map((w) => key(w.word)).filter(Boolean);
|
|
366
|
+
|
|
367
|
+
const N = 4;
|
|
368
|
+
|
|
369
|
+
/**
|
|
370
|
+
* How often the SCRIPT says each phrase.
|
|
371
|
+
*
|
|
372
|
+
* "go to the code tab" appears twice in this script on purpose, and so does "I like how the".
|
|
373
|
+
* Flagging every repeated phrase would report both as stutters. A repeat is only a fault when
|
|
374
|
+
* the cut says something more often than the script does.
|
|
375
|
+
*/
|
|
376
|
+
const scriptGrams = new Map();
|
|
377
|
+
{
|
|
378
|
+
const sk = atoms.flatMap((a) => keysOf(a));
|
|
379
|
+
for (let i = 0; i + N <= sk.length; i++) {
|
|
380
|
+
const gram = sk.slice(i, i + N).join(' ');
|
|
381
|
+
scriptGrams.set(gram, (scriptGrams.get(gram) ?? 0) + 1);
|
|
382
|
+
}
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
const repeats = [];
|
|
386
|
+
const seen = new Map();
|
|
387
|
+
for (let i = 0; i + N <= outKeys.length; i++) {
|
|
388
|
+
const gram = outKeys.slice(i, i + N).join(' ');
|
|
389
|
+
const at = outWords[i].start;
|
|
390
|
+
const previous = seen.get(gram) ?? [];
|
|
391
|
+
if (previous.length >= (scriptGrams.get(gram) ?? 0) && previous.some((t) => at - t < 20)) {
|
|
392
|
+
repeats.push({gram, first: Number(previous[previous.length - 1].toFixed(2)), again: Number(at.toFixed(2))});
|
|
393
|
+
}
|
|
394
|
+
seen.set(gram, [...previous, at]);
|
|
395
|
+
}
|
|
396
|
+
|
|
397
|
+
const survivingFiller = outWords.filter((w) => FILLER.has(key(w.word))).map((w) => ({
|
|
398
|
+
word: w.word,
|
|
399
|
+
at: Number(w.start.toFixed(2)),
|
|
400
|
+
}));
|
|
401
|
+
|
|
402
|
+
const checks = {
|
|
403
|
+
everyAtomPlaced: missing.length === 0,
|
|
404
|
+
noRepeatedPhrase: repeats.length === 0,
|
|
405
|
+
fillerLeftInside: survivingFiller.length,
|
|
406
|
+
atoms: atoms.length,
|
|
407
|
+
placed: atoms.length - missing.length,
|
|
408
|
+
keptSeconds: Number(keep.reduce((a, k) => a + (k.to - k.from), 0).toFixed(2)),
|
|
409
|
+
sourceSeconds: Number((words[words.length - 1].end - words[0].start).toFixed(2)),
|
|
410
|
+
};
|
|
411
|
+
|
|
412
|
+
/* ---------------------------------------------------------------------- output */
|
|
413
|
+
|
|
414
|
+
const stem = basename(wordsPath).replace(/\.words\.json$|\.json$/, '');
|
|
415
|
+
const project = flag('project', null);
|
|
416
|
+
const outPath =
|
|
417
|
+
flag('out', null) ??
|
|
418
|
+
(project ? join('work', project, `${stem}.takes.json`) : join(dirname(wordsPath), `${stem}.takes.json`));
|
|
419
|
+
mkdirSync(dirname(outPath), {recursive: true});
|
|
420
|
+
|
|
421
|
+
writeFileSync(
|
|
422
|
+
outPath,
|
|
423
|
+
JSON.stringify(
|
|
424
|
+
{
|
|
425
|
+
fps: FPS,
|
|
426
|
+
gapKeep: GAP_KEEP,
|
|
427
|
+
checks,
|
|
428
|
+
keep,
|
|
429
|
+
dropped,
|
|
430
|
+
repeats,
|
|
431
|
+
missing,
|
|
432
|
+
atoms: atoms.map((text, i) => ({
|
|
433
|
+
text,
|
|
434
|
+
chosen: chosen[i] ? {from: chosen[i].from, to: chosen[i].to, coverage: Number(chosen[i].coverage.toFixed(2))} : null,
|
|
435
|
+
attempts: takesPerAtom[i].length,
|
|
436
|
+
})),
|
|
437
|
+
},
|
|
438
|
+
null,
|
|
439
|
+
2,
|
|
440
|
+
),
|
|
441
|
+
);
|
|
442
|
+
|
|
443
|
+
const pct = ((checks.keptSeconds / checks.sourceSeconds) * 100).toFixed(0);
|
|
444
|
+
console.log(`${outPath}`);
|
|
445
|
+
console.log(` ${checks.placed}/${checks.atoms} script sentences placed`);
|
|
446
|
+
console.log(` ${checks.keptSeconds}s kept of ${checks.sourceSeconds}s (${pct}%)`);
|
|
447
|
+
console.log(` ${dropped.filter((d) => d.reason === 'retake').length} retakes, ${dropped.filter((d) => d.reason === 'off-script').length} off-script, ${dropped.filter((d) => d.reason === 'filler').length} filler dropped`);
|
|
448
|
+
|
|
449
|
+
let failed = false;
|
|
450
|
+
if (!checks.everyAtomPlaced) {
|
|
451
|
+
failed = true;
|
|
452
|
+
console.error(`\n${missing.length} script sentence(s) have no usable take:`);
|
|
453
|
+
for (const m of missing.slice(0, 10)) console.error(` - "${m.text.slice(0, 70)}" (${m.attempts} attempts found)`);
|
|
454
|
+
console.error(' Re-record these lines, or re-run with --loose to lower the match threshold.');
|
|
455
|
+
}
|
|
456
|
+
if (repeats.length) {
|
|
457
|
+
failed = true;
|
|
458
|
+
console.error(`\n${repeats.length} phrase(s) still repeat in the cut:`);
|
|
459
|
+
for (const r of repeats.slice(0, 10)) console.error(` - "${r.gram}" at ${r.first}s and again at ${r.again}s`);
|
|
460
|
+
}
|
|
461
|
+
if (survivingFiller.length) {
|
|
462
|
+
console.log(`\n ${survivingFiller.length} filler word(s) kept inside winning takes — left in on purpose.`);
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
if (failed && !has('loose')) process.exit(1);
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
import React from 'react';
|
|
2
|
+
import {useCurrentFrame, useVideoConfig} from 'remotion';
|
|
3
|
+
import {COLOR, FONT, PALETTE, RADIUS, SHADOW, SPACE, TYPE, WEIGHT} from '../design/tokens';
|
|
4
|
+
import {EASE} from '../design/motion';
|
|
5
|
+
import {progress} from '../lib/animate';
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* A command, typed out, as a designed graphic rather than a screen recording.
|
|
9
|
+
*
|
|
10
|
+
* Why this exists instead of zooming into the capture: the connect command carries the buyer's
|
|
11
|
+
* licence key, and the key IS the URL — anyone who can read it has a free studio. Blurring it in
|
|
12
|
+
* the capture is fragile, because the chat scrolls and the bubble moves, so a static mask misses
|
|
13
|
+
* it and a tracked one drifts. Rebuilding the command as type means there is nothing to miss: the
|
|
14
|
+
* key is never on screen in the first place, it is drawn as a redaction by design.
|
|
15
|
+
*
|
|
16
|
+
* It also reads better. A 1080p capture punched in to fill a 1920 frame is soft; this is sharp at
|
|
17
|
+
* any zoom, in the brand's own colours, and the Mac and Windows variants can sit side by side,
|
|
18
|
+
* which is impossible with one person's screen.
|
|
19
|
+
*
|
|
20
|
+
* MONOSPACE IS A DELIBERATE EXCEPTION. The design system says one family at every weight, and that
|
|
21
|
+
* holds everywhere except here: a command is code, and proportional type makes a URL ambiguous at
|
|
22
|
+
* exactly the moment someone is trying to read one. Nowhere else in the system gets this.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
export const MONO = 'ui-monospace, "SF Mono", Menlo, Consolas, "Liberation Mono", monospace';
|
|
26
|
+
|
|
27
|
+
export type Segment =
|
|
28
|
+
/** Ordinary command text. */
|
|
29
|
+
| {kind: 'plain'; text: string}
|
|
30
|
+
/** Flags and switches — dimmed, because nobody needs to read them to follow along. */
|
|
31
|
+
| {kind: 'flag'; text: string}
|
|
32
|
+
/** The part that matters, in the accent. */
|
|
33
|
+
| {kind: 'accent'; text: string}
|
|
34
|
+
/**
|
|
35
|
+
* A secret. Never pass the real value: `text` is only used for its LENGTH, so the redaction
|
|
36
|
+
* is the right width. Pass dummy characters.
|
|
37
|
+
*/
|
|
38
|
+
| {kind: 'redacted'; text: string; label?: string};
|
|
39
|
+
|
|
40
|
+
const COLOR_FOR: Record<Segment['kind'], string> = {
|
|
41
|
+
plain: PALETTE.stone100,
|
|
42
|
+
flag: PALETTE.stone400,
|
|
43
|
+
accent: COLOR.accent,
|
|
44
|
+
redacted: 'transparent',
|
|
45
|
+
};
|
|
46
|
+
|
|
47
|
+
/** How long the typing takes, from how much there is to type. Long commands do not crawl. */
|
|
48
|
+
const typeFrames = (chars: number, fps: number) => Math.round((chars / 42) * fps);
|
|
49
|
+
|
|
50
|
+
const Redaction: React.FC<{width: number; label?: string}> = ({width, label}) => (
|
|
51
|
+
<span
|
|
52
|
+
style={{
|
|
53
|
+
position: 'relative',
|
|
54
|
+
display: 'inline-flex',
|
|
55
|
+
alignItems: 'center',
|
|
56
|
+
justifyContent: 'center',
|
|
57
|
+
width,
|
|
58
|
+
height: '1.15em',
|
|
59
|
+
verticalAlign: 'middle',
|
|
60
|
+
borderRadius: RADIUS.sm / 2,
|
|
61
|
+
// Accent at low alpha, not a grey smear: it reads as a deliberate mark rather than a mistake.
|
|
62
|
+
backgroundColor: `${PALETTE.lime400}2e`,
|
|
63
|
+
border: `2px solid ${PALETTE.lime400}66`,
|
|
64
|
+
}}
|
|
65
|
+
>
|
|
66
|
+
<span
|
|
67
|
+
style={{
|
|
68
|
+
fontFamily: FONT.sans,
|
|
69
|
+
fontSize: '0.56em',
|
|
70
|
+
fontWeight: WEIGHT.semibold,
|
|
71
|
+
letterSpacing: '0.08em',
|
|
72
|
+
textTransform: 'uppercase',
|
|
73
|
+
color: COLOR.accent,
|
|
74
|
+
whiteSpace: 'nowrap',
|
|
75
|
+
}}
|
|
76
|
+
>
|
|
77
|
+
{label ?? 'your key'}
|
|
78
|
+
</span>
|
|
79
|
+
</span>
|
|
80
|
+
);
|
|
81
|
+
|
|
82
|
+
export const CommandCard: React.FC<{
|
|
83
|
+
segments: Segment[];
|
|
84
|
+
/** Frame the card lands on. Typing starts once it has settled. */
|
|
85
|
+
at: number;
|
|
86
|
+
/** Shown in the title bar. Omit for a card that is the same on both platforms. */
|
|
87
|
+
platform?: 'windows' | 'mac';
|
|
88
|
+
/** Replaces the platform chip — e.g. "same on Mac and Windows". */
|
|
89
|
+
note?: string;
|
|
90
|
+
fontSize?: number;
|
|
91
|
+
width?: number;
|
|
92
|
+
}> = ({segments, at, platform, note, fontSize = 34, width}) => {
|
|
93
|
+
const frame = useCurrentFrame();
|
|
94
|
+
const {fps} = useVideoConfig();
|
|
95
|
+
|
|
96
|
+
// The card arrives, THEN types. Typing while the card is still moving reads as two
|
|
97
|
+
// competing animations and the eye does not know where to settle.
|
|
98
|
+
const land = progress(frame, at, Math.round(0.4 * fps), EASE.out);
|
|
99
|
+
|
|
100
|
+
const total = segments.reduce((n, s) => n + s.text.length, 0);
|
|
101
|
+
const typed = progress(frame, at + Math.round(0.3 * fps), typeFrames(total, fps), EASE.inOut);
|
|
102
|
+
const shown = Math.floor(typed * total);
|
|
103
|
+
|
|
104
|
+
// A caret that blinks only once typing has finished, the way a real prompt does.
|
|
105
|
+
const done = typed >= 1;
|
|
106
|
+
const caretOn = !done || Math.floor((frame / fps) * 2) % 2 === 0;
|
|
107
|
+
|
|
108
|
+
let consumed = 0;
|
|
109
|
+
|
|
110
|
+
return (
|
|
111
|
+
<div
|
|
112
|
+
style={{
|
|
113
|
+
width,
|
|
114
|
+
maxWidth: '100%',
|
|
115
|
+
borderRadius: RADIUS.md,
|
|
116
|
+
backgroundColor: PALETTE.ink900,
|
|
117
|
+
boxShadow: SHADOW.pop,
|
|
118
|
+
overflow: 'hidden',
|
|
119
|
+
opacity: land,
|
|
120
|
+
transform: `translateY(${(1 - land) * 28}px)`,
|
|
121
|
+
}}
|
|
122
|
+
>
|
|
123
|
+
<div
|
|
124
|
+
style={{
|
|
125
|
+
display: 'flex',
|
|
126
|
+
alignItems: 'center',
|
|
127
|
+
gap: SPACE[2],
|
|
128
|
+
padding: `${SPACE[2]}px ${SPACE[3]}px`,
|
|
129
|
+
backgroundColor: PALETTE.ink800,
|
|
130
|
+
borderBottom: `2px solid ${PALETTE.ink700}`,
|
|
131
|
+
}}
|
|
132
|
+
>
|
|
133
|
+
{[PALETTE.stone600, PALETTE.stone600, PALETTE.stone600].map((c, i) => (
|
|
134
|
+
<span key={i} style={{width: 14, height: 14, borderRadius: RADIUS.pill, backgroundColor: c}} />
|
|
135
|
+
))}
|
|
136
|
+
<span style={{flex: 1}} />
|
|
137
|
+
<span
|
|
138
|
+
style={{
|
|
139
|
+
fontFamily: FONT.sans,
|
|
140
|
+
fontSize: TYPE.eyebrow.size * 0.62,
|
|
141
|
+
fontWeight: WEIGHT.medium,
|
|
142
|
+
letterSpacing: TYPE.eyebrow.tracking,
|
|
143
|
+
textTransform: 'uppercase',
|
|
144
|
+
color: note ? COLOR.accent : PALETTE.stone400,
|
|
145
|
+
}}
|
|
146
|
+
>
|
|
147
|
+
{note ?? (platform === 'mac' ? 'macOS' : platform === 'windows' ? 'Windows' : '')}
|
|
148
|
+
</span>
|
|
149
|
+
</div>
|
|
150
|
+
|
|
151
|
+
<div
|
|
152
|
+
style={{
|
|
153
|
+
padding: `${SPACE[4]}px ${SPACE[4]}px ${SPACE[4]}px`,
|
|
154
|
+
fontFamily: MONO,
|
|
155
|
+
fontSize,
|
|
156
|
+
lineHeight: 1.65,
|
|
157
|
+
letterSpacing: '-0.01em',
|
|
158
|
+
color: PALETTE.stone100,
|
|
159
|
+
wordBreak: 'break-word',
|
|
160
|
+
}}
|
|
161
|
+
>
|
|
162
|
+
{segments.map((seg, i) => {
|
|
163
|
+
const start = consumed;
|
|
164
|
+
consumed += seg.text.length;
|
|
165
|
+
const visible = Math.max(0, Math.min(seg.text.length, shown - start));
|
|
166
|
+
if (visible === 0) return null;
|
|
167
|
+
|
|
168
|
+
if (seg.kind === 'redacted') {
|
|
169
|
+
// Reveal the redaction as one block once typing reaches it, rather than growing it
|
|
170
|
+
// character by character, which would look like the secret is being drawn in.
|
|
171
|
+
const w = seg.text.length * fontSize * 0.6;
|
|
172
|
+
return <Redaction key={i} width={w} label={seg.label} />;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
return (
|
|
176
|
+
<span key={i} style={{color: COLOR_FOR[seg.kind]}}>
|
|
177
|
+
{seg.text.slice(0, visible)}
|
|
178
|
+
</span>
|
|
179
|
+
);
|
|
180
|
+
})}
|
|
181
|
+
<span
|
|
182
|
+
style={{
|
|
183
|
+
display: 'inline-block',
|
|
184
|
+
width: fontSize * 0.55,
|
|
185
|
+
height: '1.05em',
|
|
186
|
+
marginLeft: 4,
|
|
187
|
+
verticalAlign: 'text-bottom',
|
|
188
|
+
backgroundColor: COLOR.accent,
|
|
189
|
+
opacity: caretOn ? 1 : 0,
|
|
190
|
+
}}
|
|
191
|
+
/>
|
|
192
|
+
</div>
|
|
193
|
+
</div>
|
|
194
|
+
);
|
|
195
|
+
};
|
|
196
|
+
|
|
197
|
+
/* ------------------------------------------------------------------ the real commands */
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* The connect line. Identical on Mac and Windows, which is worth SAYING rather than showing
|
|
201
|
+
* twice: a Mac viewer watching a Windows walkthrough is waiting to find out what is different
|
|
202
|
+
* for them, and "nothing, here" is the answer that keeps them watching.
|
|
203
|
+
*/
|
|
204
|
+
export const CONNECT_SEGMENTS: Segment[] = [
|
|
205
|
+
{kind: 'plain', text: 'claude mcp add '},
|
|
206
|
+
{kind: 'flag', text: '--scope user --transport http '},
|
|
207
|
+
{kind: 'accent', text: 'creator-editing-os'},
|
|
208
|
+
{kind: 'plain', text: ' https://creator-editing-os.mahipal-231.workers.dev/mcp/'},
|
|
209
|
+
{kind: 'redacted', text: 'XXXX-XXXX-XXXX-XXXX'},
|
|
210
|
+
];
|
|
211
|
+
|
|
212
|
+
/** Where the two platforms genuinely differ: installing Node when it is missing. */
|
|
213
|
+
export const NODE_INSTALL = {
|
|
214
|
+
windows: [
|
|
215
|
+
{kind: 'plain', text: 'winget install '},
|
|
216
|
+
{kind: 'accent', text: 'OpenJS.NodeJS.LTS'},
|
|
217
|
+
] as Segment[],
|
|
218
|
+
mac: [
|
|
219
|
+
{kind: 'plain', text: 'brew install '},
|
|
220
|
+
{kind: 'accent', text: 'node'},
|
|
221
|
+
] as Segment[],
|
|
222
|
+
};
|
|
223
|
+
|
|
224
|
+
/** And where the studio lands. */
|
|
225
|
+
export const STUDIO_PATH = {
|
|
226
|
+
windows: [{kind: 'plain', text: 'C:\\Users\\you\\'}, {kind: 'accent', text: 'VideoStudio'}] as Segment[],
|
|
227
|
+
mac: [{kind: 'plain', text: '/Users/you/'}, {kind: 'accent', text: 'VideoStudio'}] as Segment[],
|
|
228
|
+
};
|