creator-editing-studio 1.2.1 → 1.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/template/scripts/takes.mjs +336 -377
package/package.json
CHANGED
|
@@ -1,35 +1,38 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
/**
|
|
3
|
-
* Best-take selection:
|
|
3
|
+
* Best-take selection: one messy take in, one clean cut plan out.
|
|
4
4
|
*
|
|
5
|
-
* npm run takes -- <
|
|
6
|
-
* [--
|
|
5
|
+
* npm run takes -- <words.json> [--media <video>] [--script <script.md>]
|
|
6
|
+
* [--project <name>] [--out <file>] [--fps 30]
|
|
7
|
+
* [--similar 0.72] [--window 90] [--gap-max 0.6] [--gap-keep 0.25]
|
|
7
8
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
* trip to CapCut to take out repeats nobody caught.
|
|
9
|
+
* WHAT IT IS FOR. A talking head is full of second attempts. Somebody fumbles, stops, says the
|
|
10
|
+
* sentence again, and both versions end up in the cut — so a human watches the whole thing to
|
|
11
|
+
* find them. This finds them instead, keeps the better one, and proves afterwards that none are
|
|
12
|
+
* left.
|
|
13
13
|
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
14
|
+
* THE SPEECH IS THE TRUTH, NOT THE SCRIPT. An earlier version of this file anchored to the
|
|
15
|
+
* written script: every script sentence got one span and anything unmatched was dropped. It
|
|
16
|
+
* threw away six passages of real material on the first take it was pointed at — lines the
|
|
17
|
+
* speaker improvised on the day and meant to keep, including "you don't have to put the brief
|
|
18
|
+
* into the brief file". A repeat that survives is visible and fixable; a good line silently
|
|
19
|
+
* deleted is neither. So the default here is inverted:
|
|
19
20
|
*
|
|
20
|
-
*
|
|
21
|
-
* line, instead of handing over a cut with a hole for someone to find on playback.
|
|
21
|
+
* KEEP EVERYTHING, UNLESS IT IS A SECOND ATTEMPT AT SOMETHING ELSE THAT IS KEPT.
|
|
22
22
|
*
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
* atoms[] every script sentence, its chosen take and the attempts that lost
|
|
27
|
-
* checks{} the verification results, all of which must pass
|
|
23
|
+
* Nothing has to justify its existence. A script, if there is one, is only a checker: it reports
|
|
24
|
+
* lines that never got said and breaks ties between close takes. It can never remove anything,
|
|
25
|
+
* and most people filming to camera do not have one.
|
|
28
26
|
*
|
|
29
|
-
*
|
|
27
|
+
* WHAT IT WRITES work/<project>/<stem>.takes.json
|
|
28
|
+
* keep[] {from, to, units} source seconds to use, in order
|
|
29
|
+
* groups[] every retake cluster, which attempt won and why the others lost
|
|
30
|
+
* dropped[] what went, with the reason
|
|
31
|
+
* checks{} verification, all of which must pass
|
|
30
32
|
*/
|
|
31
33
|
import {existsSync, mkdirSync, readFileSync, writeFileSync} from 'node:fs';
|
|
32
34
|
import {basename, dirname, join} from 'node:path';
|
|
35
|
+
import {decodePcm} from './lib/media.mjs';
|
|
33
36
|
|
|
34
37
|
/* ------------------------------------------------------------------------ args */
|
|
35
38
|
|
|
@@ -38,292 +41,304 @@ const flag = (name, fallback) => {
|
|
|
38
41
|
const i = args.indexOf(`--${name}`);
|
|
39
42
|
return i === -1 ? fallback : args[i + 1];
|
|
40
43
|
};
|
|
41
|
-
const has = (name) => args.includes(`--${name}`);
|
|
42
44
|
const positional = args.filter((a, i) => !a.startsWith('--') && !String(args[i - 1] ?? '').startsWith('--'));
|
|
43
45
|
|
|
44
|
-
const
|
|
45
|
-
if (!
|
|
46
|
-
console.error('usage: npm run takes -- <
|
|
46
|
+
const wordsPath = positional[0];
|
|
47
|
+
if (!wordsPath) {
|
|
48
|
+
console.error('usage: npm run takes -- <words.json> [--media <video>] [--script <script.md>]');
|
|
47
49
|
process.exit(1);
|
|
48
50
|
}
|
|
49
51
|
|
|
50
52
|
const FPS = Number(flag('fps', 30));
|
|
51
|
-
/**
|
|
52
|
-
const
|
|
53
|
-
/**
|
|
54
|
-
const
|
|
55
|
-
/**
|
|
56
|
-
const
|
|
53
|
+
/** How alike two passages must be before they count as attempts at the same thing. */
|
|
54
|
+
const SIMILAR = Number(flag('similar', 0.72));
|
|
55
|
+
/** Retakes happen close together. Two matching sentences minutes apart are both wanted. */
|
|
56
|
+
const WINDOW = Number(flag('window', 90));
|
|
57
|
+
/** A silence longer than this is dead air and gets collapsed. */
|
|
58
|
+
const GAP_MAX = Number(flag('gap-max', 0.6));
|
|
59
|
+
/** ...down to this. Not zero: speech with every gap removed sounds like a machine. */
|
|
60
|
+
const GAP_KEEP = Number(flag('gap-keep', 0.25));
|
|
61
|
+
/**
|
|
62
|
+
* Breathing room kept either side of a kept passage.
|
|
63
|
+
*
|
|
64
|
+
* 150ms, not the 60 an earlier version used. Below about 120 the cut lands close enough to the
|
|
65
|
+
* first phoneme to clip it, and the join reads as abrupt even when every word is intact.
|
|
66
|
+
*/
|
|
67
|
+
const PAD = 0.15;
|
|
57
68
|
|
|
58
|
-
/*
|
|
69
|
+
/* ------------------------------------------------------------------- the words */
|
|
70
|
+
|
|
71
|
+
const key = (t) => t.toLowerCase().replace(/['‘’]/g, '').replace(/[^a-z0-9]/g, '');
|
|
72
|
+
|
|
73
|
+
const words = JSON.parse(readFileSync(wordsPath, 'utf8')).words.filter((w) => w.start != null && w.end != null);
|
|
74
|
+
if (words.length < 10) {
|
|
75
|
+
console.error('Not enough timed words to work with.');
|
|
76
|
+
process.exit(1);
|
|
77
|
+
}
|
|
59
78
|
|
|
60
79
|
/**
|
|
61
|
-
*
|
|
80
|
+
* Cues that announce a retake out loud.
|
|
62
81
|
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
* blocks are stripped — they are instructions to the writer, never spoken.
|
|
82
|
+
* When someone says one of these, the passage BEFORE it was abandoned — they are telling you so.
|
|
83
|
+
* Cheaper and more certain than any similarity measure.
|
|
66
84
|
*/
|
|
67
|
-
const
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
85
|
+
const RESTART_CUES = [
|
|
86
|
+
'let me redo that', 'let me try that again', 'one more time', 'sorry let me',
|
|
87
|
+
'can i do that again', 'take two', 'scratch that', 'let me say that again',
|
|
88
|
+
'start again', 'i will say that again',
|
|
89
|
+
];
|
|
90
|
+
|
|
91
|
+
/* --------------------------------------------------------------- units of speech
|
|
92
|
+
* A unit is one spoken thought: it ends where the speaker put a full stop, or where they left a
|
|
93
|
+
* gap long enough to be a boundary rather than a breath. Working in units rather than fixed word
|
|
94
|
+
* counts means a retake is compared against the whole thought it replaced. */
|
|
95
|
+
|
|
96
|
+
const units = [];
|
|
97
|
+
{
|
|
98
|
+
let start = 0;
|
|
99
|
+
for (let i = 0; i < words.length; i++) {
|
|
100
|
+
const w = words[i];
|
|
101
|
+
const next = words[i + 1];
|
|
102
|
+
const gapAfter = next ? next.start - w.end : Infinity;
|
|
103
|
+
const endsSentence = /[.?!]$/.test(w.word);
|
|
104
|
+
if (endsSentence || gapAfter > 0.7 || i === words.length - 1) {
|
|
105
|
+
const span = words.slice(start, i + 1);
|
|
106
|
+
if (span.length) {
|
|
107
|
+
units.push({
|
|
108
|
+
index: units.length,
|
|
109
|
+
startIndex: start,
|
|
110
|
+
endIndex: i,
|
|
111
|
+
from: span[0].start,
|
|
112
|
+
to: span[span.length - 1].end,
|
|
113
|
+
text: span.map((x) => x.word).join(' '),
|
|
114
|
+
keys: span.map((x) => key(x.word)).filter(Boolean),
|
|
115
|
+
confidence: span.reduce((a, x) => a + (x.score ?? 0.8), 0) / span.length,
|
|
116
|
+
complete: endsSentence,
|
|
117
|
+
worstGap: span.slice(1).reduce((m, x, k) => Math.max(m, x.start - span[k].end), 0),
|
|
118
|
+
});
|
|
119
|
+
}
|
|
120
|
+
start = i + 1;
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/* -------------------------------------------------------------- energy and pace */
|
|
81
126
|
|
|
82
|
-
/** Comparable key for a word. Mirrors align_script.py so both agree on what "same" means. */
|
|
83
127
|
/**
|
|
84
|
-
*
|
|
128
|
+
* Loudness per unit, so a flat read can lose to a committed one.
|
|
85
129
|
*
|
|
86
|
-
*
|
|
87
|
-
*
|
|
88
|
-
* the audio for it is thrown away as off-script. That is a worse failure than a repeat.
|
|
130
|
+
* Optional: without --media there is no audio to measure and the score simply does without it,
|
|
131
|
+
* rather than the tool refusing to run.
|
|
89
132
|
*/
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
133
|
+
let rmsOf = () => null;
|
|
134
|
+
const mediaPath = flag('media', null);
|
|
135
|
+
if (mediaPath && existsSync(mediaPath)) {
|
|
136
|
+
const RATE = 22050;
|
|
137
|
+
const pcm = decodePcm(mediaPath, RATE);
|
|
138
|
+
if (pcm) {
|
|
139
|
+
rmsOf = (from, to) => {
|
|
140
|
+
const a = Math.max(0, Math.floor(from * RATE));
|
|
141
|
+
const b = Math.min(pcm.length, Math.ceil(to * RATE));
|
|
142
|
+
if (b <= a) return null;
|
|
143
|
+
let sum = 0;
|
|
144
|
+
for (let i = a; i < b; i++) sum += pcm[i] * pcm[i];
|
|
145
|
+
return Math.sqrt(sum / (b - a));
|
|
146
|
+
};
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
for (const u of units) {
|
|
151
|
+
u.seconds = u.to - u.from;
|
|
152
|
+
u.wordsPerSecond = u.keys.length / Math.max(0.3, u.seconds);
|
|
153
|
+
u.rms = rmsOf(u.from, u.to);
|
|
154
|
+
}
|
|
94
155
|
|
|
95
|
-
const
|
|
96
|
-
const
|
|
97
|
-
return
|
|
156
|
+
const median = (xs) => {
|
|
157
|
+
const s = xs.filter((x) => x != null).sort((a, b) => a - b);
|
|
158
|
+
return s.length ? s[Math.floor(s.length / 2)] : null;
|
|
159
|
+
};
|
|
160
|
+
const medianRms = median(units.map((u) => u.rms));
|
|
161
|
+
const medianPace = median(units.map((u) => u.wordsPerSecond)) ?? 3;
|
|
162
|
+
|
|
163
|
+
/* ------------------------------------------------------------------ similarity */
|
|
164
|
+
|
|
165
|
+
/** Longest common subsequence length — tolerant of an inserted stumble or a swapped word. */
|
|
166
|
+
const lcs = (a, b) => {
|
|
167
|
+
const prev = new Array(b.length + 1).fill(0);
|
|
168
|
+
const cur = new Array(b.length + 1).fill(0);
|
|
169
|
+
for (let i = 1; i <= a.length; i++) {
|
|
170
|
+
for (let j = 1; j <= b.length; j++) {
|
|
171
|
+
cur[j] = a[i - 1] === b[j - 1] ? prev[j - 1] + 1 : Math.max(prev[j], cur[j - 1]);
|
|
172
|
+
}
|
|
173
|
+
for (let j = 0; j <= b.length; j++) prev[j] = cur[j];
|
|
174
|
+
}
|
|
175
|
+
return prev[b.length];
|
|
98
176
|
};
|
|
99
177
|
|
|
100
178
|
/**
|
|
101
|
-
*
|
|
179
|
+
* How alike two units are, measured against the SHORTER one.
|
|
102
180
|
*
|
|
103
|
-
*
|
|
104
|
-
*
|
|
105
|
-
*
|
|
181
|
+
* Against the shorter, because an abandoned start — "so the main reason, so the main reason is
|
|
182
|
+
* that…" — is a short fragment entirely contained in the full attempt. Measured against the
|
|
183
|
+
* longer it would score low and be kept; against the shorter it scores 1.0 and is correctly
|
|
184
|
+
* recognised as the same thought, begun twice.
|
|
106
185
|
*/
|
|
107
|
-
const
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
/** One spoken or written word becomes the one or more keys it is equivalent to. */
|
|
119
|
-
const keysOfWord = (token) => {
|
|
120
|
-
const k = key(token);
|
|
121
|
-
if (!k) return [];
|
|
122
|
-
// Only expand when the original actually had an apostrophe, so ordinary words are left alone.
|
|
123
|
-
if (/['‘’]/.test(token) && CONTRACTION.has(k)) return CONTRACTION.get(k);
|
|
124
|
-
return [k];
|
|
186
|
+
const similarity = (a, b) => {
|
|
187
|
+
if (!a.keys.length || !b.keys.length) return 0;
|
|
188
|
+
const shorter = Math.min(a.keys.length, b.keys.length);
|
|
189
|
+
// Short units match each other on function words alone — "give it three things" against
|
|
190
|
+
// "this is the time you give it that" shares "give it" and scores 0.5 on a four-word unit.
|
|
191
|
+
// Below five words there is not enough signal to call anything a retake.
|
|
192
|
+
if (shorter < 5) return 0;
|
|
193
|
+
const overlap = lcs(a.keys, b.keys);
|
|
194
|
+
// And the overlap has to be real words, not two articles and a pronoun.
|
|
195
|
+
if (overlap < 4) return 0;
|
|
196
|
+
return overlap / shorter;
|
|
125
197
|
};
|
|
126
198
|
|
|
127
|
-
|
|
199
|
+
/* ---------------------------------------------------------------------- groups */
|
|
200
|
+
|
|
201
|
+
const groupOf = new Array(units.length).fill(-1);
|
|
202
|
+
const groups = [];
|
|
128
203
|
|
|
129
204
|
/**
|
|
130
|
-
*
|
|
205
|
+
* Grouping is against a representative, never transitive.
|
|
131
206
|
*
|
|
132
|
-
*
|
|
133
|
-
*
|
|
134
|
-
*
|
|
135
|
-
*
|
|
207
|
+
* Chaining members together — A matches B, B matches C, so group them all — merges passages that
|
|
208
|
+
* have nothing to do with each other as soon as one ambiguous unit sits between them. It put
|
|
209
|
+
* "give it three things" in the same group as "they just won't really feel like you". A unit
|
|
210
|
+
* joins a group only if it is similar to that group's FIRST member, which keeps a group to one
|
|
211
|
+
* thought said more than once.
|
|
136
212
|
*/
|
|
137
|
-
|
|
213
|
+
for (let i = 0; i < units.length; i++) {
|
|
214
|
+
if (groupOf[i] !== -1) continue;
|
|
215
|
+
const members = [i];
|
|
216
|
+
for (let j = i + 1; j < units.length; j++) {
|
|
217
|
+
if (groupOf[j] !== -1) continue;
|
|
218
|
+
if (units[j].from - units[i].to > WINDOW) break;
|
|
219
|
+
if (similarity(units[i], units[j]) < SIMILAR) continue;
|
|
220
|
+
members.push(j);
|
|
221
|
+
}
|
|
222
|
+
if (members.length < 2) continue;
|
|
223
|
+
groups.push({members});
|
|
224
|
+
for (const m of members) groupOf[m] = groups.length - 1;
|
|
225
|
+
}
|
|
138
226
|
|
|
139
|
-
|
|
140
|
-
const
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
const ks = keysOfWord(w.word);
|
|
147
|
-
return ks.length ? ks : ['\u0000'];
|
|
148
|
-
});
|
|
149
|
-
const tokens = tokenKeys.map((ks) => ks[0]);
|
|
150
|
-
|
|
151
|
-
if (!atoms.length) {
|
|
152
|
-
console.error(`No script sentences found in ${scriptPath}.`);
|
|
153
|
-
process.exit(1);
|
|
227
|
+
/** A spoken cue abandons whatever came immediately before it. */
|
|
228
|
+
const abandoned = new Set();
|
|
229
|
+
for (const u of units) {
|
|
230
|
+
const flat = u.keys.join(' ');
|
|
231
|
+
if (!RESTART_CUES.some((c) => flat.includes(c.replace(/[^a-z0-9 ]/g, '')))) continue;
|
|
232
|
+
const prev = units[u.index - 1];
|
|
233
|
+
if (prev) abandoned.add(prev.index);
|
|
154
234
|
}
|
|
155
235
|
|
|
156
|
-
/*
|
|
236
|
+
/* --------------------------------------------------------------------- scoring */
|
|
157
237
|
|
|
158
238
|
/**
|
|
159
|
-
*
|
|
239
|
+
* One number per attempt.
|
|
160
240
|
*
|
|
161
|
-
*
|
|
162
|
-
*
|
|
163
|
-
*
|
|
164
|
-
* an "um" inside a sentence is exactly what a fumbled attempt sounds like.
|
|
241
|
+
* Completeness dominates — a finished sentence beats a smoother fragment every time. Everything
|
|
242
|
+
* else nudges. The tie-break at the end is deliberate and matches how people actually record:
|
|
243
|
+
* they repeat a line until they are happy, so the last attempt is usually the keeper.
|
|
165
244
|
*/
|
|
166
|
-
const
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
* line is reported missing while its audio is discarded. Coverage still has to clear the
|
|
179
|
-
* threshold, so a take that drops its opening words scores lower rather than passing free.
|
|
180
|
-
*/
|
|
181
|
-
const anchors = want.slice(0, 3);
|
|
182
|
-
|
|
183
|
-
for (let start = 0; start < tokens.length; start++) {
|
|
184
|
-
const anchorAt = anchors.indexOf(tokens[start]);
|
|
185
|
-
if (anchorAt === -1) continue;
|
|
186
|
-
|
|
187
|
-
let wi = anchorAt;
|
|
188
|
-
let matched = anchorAt;
|
|
189
|
-
let last = start;
|
|
190
|
-
let extras = 0;
|
|
191
|
-
for (let ti = start; ti < Math.min(tokens.length, start + window) && wi < want.length; ti++) {
|
|
192
|
-
// A contraction covers more than one script word, so consume as many as it supplies.
|
|
193
|
-
let consumed = 0;
|
|
194
|
-
while (consumed < tokenKeys[ti].length && wi + consumed < want.length && tokenKeys[ti][consumed] === want[wi + consumed]) consumed++;
|
|
195
|
-
if (consumed > 0) {
|
|
196
|
-
matched += consumed;
|
|
197
|
-
wi += consumed;
|
|
198
|
-
last = ti;
|
|
199
|
-
} else if (want.indexOf(tokens[ti], wi) !== -1 && want.indexOf(tokens[ti], wi) - wi <= 3) {
|
|
200
|
-
// The speaker skipped a word or two and carried on; follow them.
|
|
201
|
-
wi = want.indexOf(tokens[ti], wi) + 1;
|
|
202
|
-
matched++;
|
|
203
|
-
last = ti;
|
|
204
|
-
} else {
|
|
205
|
-
extras++;
|
|
206
|
-
}
|
|
207
|
-
}
|
|
208
|
-
|
|
209
|
-
const coverage = matched / want.length;
|
|
210
|
-
if (coverage < MIN_COVERAGE) continue;
|
|
211
|
-
|
|
212
|
-
const from = words[start].start;
|
|
213
|
-
const to = words[last].end;
|
|
214
|
-
const span = words.slice(start, last + 1);
|
|
215
|
-
|
|
216
|
-
let worstGap = 0;
|
|
217
|
-
for (let i = 1; i < span.length; i++) worstGap = Math.max(worstGap, span[i].start - span[i - 1].end);
|
|
218
|
-
const fillerInside = span.filter((w) => FILLER.has(key(w.word))).length;
|
|
219
|
-
const confidence = span.reduce((a, w) => a + (w.score ?? 0.8), 0) / span.length;
|
|
220
|
-
|
|
221
|
-
out.push({
|
|
222
|
-
atomIndex,
|
|
223
|
-
startIndex: start,
|
|
224
|
-
endIndex: last,
|
|
225
|
-
from,
|
|
226
|
-
to,
|
|
227
|
-
coverage,
|
|
228
|
-
confidence,
|
|
229
|
-
worstGap,
|
|
230
|
-
fillerInside,
|
|
231
|
-
extras,
|
|
232
|
-
text: span.map((w) => w.word).join(' '),
|
|
233
|
-
/**
|
|
234
|
-
* One number, so the chooser has something to maximise.
|
|
235
|
-
*
|
|
236
|
-
* Coverage dominates: a complete sentence beats a smooth fragment every time. The small
|
|
237
|
-
* bonus for starting later is there because when someone fumbles they fix it on the next
|
|
238
|
-
* try, so with two otherwise equal attempts the second is the keeper.
|
|
239
|
-
*/
|
|
240
|
-
score:
|
|
241
|
-
coverage * 10 +
|
|
242
|
-
confidence * 2 -
|
|
243
|
-
Math.max(0, worstGap - 0.4) * 3 -
|
|
244
|
-
fillerInside * 0.8 -
|
|
245
|
-
(extras / want.length) * 1.5 +
|
|
246
|
-
(start / tokens.length) * 0.3,
|
|
247
|
-
});
|
|
248
|
-
|
|
249
|
-
start = last; // attempts cannot overlap each other
|
|
250
|
-
}
|
|
251
|
-
return out;
|
|
245
|
+
const score = (u) => {
|
|
246
|
+
let s = 0;
|
|
247
|
+
s += u.complete ? 6 : 0;
|
|
248
|
+
s += Math.min(u.keys.length / 12, 1) * 3;
|
|
249
|
+
s += u.confidence * 2;
|
|
250
|
+
s -= Math.max(0, u.worstGap - 0.45) * 3;
|
|
251
|
+
if (abandoned.has(u.index)) s -= 20;
|
|
252
|
+
// Pace: too fast reads as rushed, too slow as laboured. Either direction costs the same.
|
|
253
|
+
s -= Math.min(2, Math.abs(u.wordsPerSecond - medianPace) / Math.max(0.8, medianPace) * 2);
|
|
254
|
+
// Energy, only when there was audio to measure.
|
|
255
|
+
if (u.rms != null && medianRms) s += Math.max(-1.5, Math.min(1.5, (u.rms / medianRms - 1) * 2));
|
|
256
|
+
return s;
|
|
252
257
|
};
|
|
253
258
|
|
|
254
|
-
const
|
|
259
|
+
for (const u of units) u.score = score(u);
|
|
255
260
|
|
|
256
|
-
/*
|
|
261
|
+
/* ------------------------------------------------------- optional script checker
|
|
262
|
+
* A script never removes anything. It reports lines that were never said, and nudges a tie
|
|
263
|
+
* toward whichever attempt is closest to what was written. */
|
|
257
264
|
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
const takes = takesPerAtom[a];
|
|
271
|
-
for (let c = 0; c < takes.length; c++) {
|
|
272
|
-
if (a === 0) {
|
|
273
|
-
best[a][c] = {total: takes[c].score, prev: -1};
|
|
274
|
-
continue;
|
|
275
|
-
}
|
|
276
|
-
let bestPrev = NEG;
|
|
277
|
-
let bestPrevIndex = -1;
|
|
278
|
-
for (let p = 0; p < takesPerAtom[a - 1].length; p++) {
|
|
279
|
-
if (best[a - 1][p].total === NEG) continue;
|
|
280
|
-
if (takesPerAtom[a - 1][p].endIndex >= takes[c].startIndex) continue;
|
|
281
|
-
if (best[a - 1][p].total > bestPrev) {
|
|
282
|
-
bestPrev = best[a - 1][p].total;
|
|
283
|
-
bestPrevIndex = p;
|
|
284
|
-
}
|
|
285
|
-
}
|
|
286
|
-
// An atom with no reachable predecessor can still start a run; the coverage check below
|
|
287
|
-
// is what decides whether the result is acceptable, not this.
|
|
288
|
-
best[a][c] =
|
|
289
|
-
bestPrevIndex === -1
|
|
290
|
-
? {total: takes[c].score, prev: -1}
|
|
291
|
-
: {total: bestPrev + takes[c].score, prev: bestPrevIndex};
|
|
292
|
-
}
|
|
293
|
-
}
|
|
265
|
+
const scriptPath = flag('script', null);
|
|
266
|
+
let scriptMissing = [];
|
|
267
|
+
if (scriptPath && existsSync(scriptPath)) {
|
|
268
|
+
const raw = readFileSync(scriptPath, 'utf8');
|
|
269
|
+
const body = /^---\s*$/m.test(raw) ? raw.split(/^---\s*$/m).slice(1).join('\n') : raw;
|
|
270
|
+
const lines = body
|
|
271
|
+
.replace(/<!--[\s\S]*?-->/g, '')
|
|
272
|
+
.replace(/^#.*$/gm, '')
|
|
273
|
+
.replace(/`[^`]*`/g, ' ')
|
|
274
|
+
.split(/(?<=[.?!])\s+/)
|
|
275
|
+
.map((s) => s.replace(/\s+/g, ' ').trim())
|
|
276
|
+
.filter((s) => s.split(/\s+/).length >= 4);
|
|
294
277
|
|
|
295
|
-
const
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
}
|
|
278
|
+
for (const line of lines) {
|
|
279
|
+
const lk = line.split(/\s+/).map(key).filter(Boolean);
|
|
280
|
+
const best = Math.max(0, ...units.map((u) => lcs(lk, u.keys) / Math.max(1, lk.length)));
|
|
281
|
+
if (best < 0.6) scriptMissing.push({text: line, bestMatch: Number(best.toFixed(2))});
|
|
282
|
+
}
|
|
283
|
+
// Closest-to-script wins a close call, but only by a little.
|
|
284
|
+
for (const g of groups) {
|
|
285
|
+
for (const m of g.members) {
|
|
286
|
+
const u = units[m];
|
|
287
|
+
const best = Math.max(0, ...lines.map((l) => lcs(l.split(/\s+/).map(key).filter(Boolean), u.keys) / Math.max(1, u.keys.length)));
|
|
288
|
+
u.score += best * 1.2;
|
|
289
|
+
}
|
|
308
290
|
}
|
|
309
|
-
if (pick === -1 || !takes[pick]) continue;
|
|
310
|
-
chosen[a] = takes[pick];
|
|
311
|
-
cursor = best[a][pick].prev;
|
|
312
291
|
}
|
|
313
292
|
|
|
314
|
-
/*
|
|
293
|
+
/* ----------------------------------------------------------------- the decision */
|
|
315
294
|
|
|
316
|
-
const
|
|
317
|
-
|
|
318
|
-
if (!chosen[i]) missing.push({index: i, text: atom, attempts: takesPerAtom[i].length});
|
|
319
|
-
});
|
|
295
|
+
const drop = new Set();
|
|
296
|
+
const groupReport = [];
|
|
320
297
|
|
|
321
|
-
const
|
|
322
|
-
.
|
|
323
|
-
|
|
324
|
-
|
|
298
|
+
for (const g of groups) {
|
|
299
|
+
const members = [...new Set(g.members)].sort((a, b) => a - b);
|
|
300
|
+
let winner = members[0];
|
|
301
|
+
for (const m of members) {
|
|
302
|
+
const a = units[m];
|
|
303
|
+
const b = units[winner];
|
|
304
|
+
// Within a hair of each other, the later one wins.
|
|
305
|
+
if (a.score > b.score + 0.3 || (Math.abs(a.score - b.score) <= 0.3 && a.from > b.from)) winner = m;
|
|
306
|
+
}
|
|
307
|
+
for (const m of members) if (m !== winner) drop.add(m);
|
|
308
|
+
groupReport.push({
|
|
309
|
+
kept: {at: Number(units[winner].from.toFixed(2)), text: units[winner].text.slice(0, 90), score: Number(units[winner].score.toFixed(2))},
|
|
310
|
+
dropped: members.filter((m) => m !== winner).map((m) => ({
|
|
311
|
+
at: Number(units[m].from.toFixed(2)),
|
|
312
|
+
text: units[m].text.slice(0, 90),
|
|
313
|
+
score: Number(units[m].score.toFixed(2)),
|
|
314
|
+
})),
|
|
315
|
+
});
|
|
316
|
+
}
|
|
317
|
+
for (const i of abandoned) drop.add(i);
|
|
318
|
+
|
|
319
|
+
const kept = units.filter((u) => !drop.has(u.index));
|
|
320
|
+
|
|
321
|
+
/* ------------------------------------------------------------- building the plan
|
|
322
|
+
* Consecutive survivors become one span, so the cut is not chopped at every sentence. A gap
|
|
323
|
+
* between spans is collapsed only when it is longer than a breath — and a pause in front of a
|
|
324
|
+
* short line is usually deliberate, so those are left alone. */
|
|
325
|
+
|
|
326
|
+
const keep = [];
|
|
327
|
+
for (const u of kept) {
|
|
328
|
+
const last = keep[keep.length - 1];
|
|
329
|
+
if (last && u.startIndex === last.endIndex + 1 && u.from - last.to < GAP_MAX) {
|
|
330
|
+
last.to = u.to;
|
|
331
|
+
last.endIndex = u.endIndex;
|
|
332
|
+
last.units.push(u.index);
|
|
333
|
+
} else {
|
|
334
|
+
keep.push({from: u.from, to: u.to, startIndex: u.startIndex, endIndex: u.endIndex, units: [u.index]});
|
|
335
|
+
}
|
|
336
|
+
}
|
|
325
337
|
|
|
326
|
-
|
|
338
|
+
for (const k of keep) {
|
|
339
|
+
k.from = Math.max(0, k.from - PAD);
|
|
340
|
+
k.to += PAD;
|
|
341
|
+
}
|
|
327
342
|
for (let i = 1; i < keep.length; i++) {
|
|
328
343
|
if (keep[i].from < keep[i - 1].to) {
|
|
329
344
|
const mid = (keep[i].from + keep[i - 1].to) / 2;
|
|
@@ -332,134 +347,78 @@ for (let i = 1; i < keep.length; i++) {
|
|
|
332
347
|
}
|
|
333
348
|
}
|
|
334
349
|
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
const
|
|
341
|
-
|
|
342
|
-
for (let i = 0; i < words.length; i++) {
|
|
343
|
-
const inCut = keptIndexes.has(i);
|
|
344
|
-
if (!inCut && runStart === null) runStart = i;
|
|
345
|
-
if ((inCut || i === words.length - 1) && runStart !== null) {
|
|
346
|
-
const end = inCut ? i - 1 : i;
|
|
347
|
-
const text = words.slice(runStart, end + 1).map((w) => w.word).join(' ');
|
|
348
|
-
const alsoKept = chosen.filter(Boolean).some((t) => keysOf(t.text).join(' ').includes(keysOf(text).join(' ')));
|
|
349
|
-
dropped.push({
|
|
350
|
-
from: words[runStart].start,
|
|
351
|
-
to: words[end].end,
|
|
352
|
-
reason: alsoKept ? 'retake' : text.split(/\s+/).length <= 2 ? 'filler' : 'off-script',
|
|
353
|
-
text,
|
|
354
|
-
});
|
|
355
|
-
runStart = null;
|
|
356
|
-
}
|
|
350
|
+
/** Gaps between spans, after the cut: long dead air collapsed, deliberate beats kept. */
|
|
351
|
+
const gaps = [];
|
|
352
|
+
for (let i = 1; i < keep.length; i++) {
|
|
353
|
+
const gap = keep[i].from - keep[i - 1].to;
|
|
354
|
+
const nextIsShort = units[keep[i].units[0]].keys.length <= 6;
|
|
355
|
+
const target = gap > GAP_MAX && !nextIsShort ? GAP_KEEP : gap;
|
|
356
|
+
gaps.push({after: i - 1, was: Number(gap.toFixed(2)), becomes: Number(target.toFixed(2)), keptAsBeat: target === gap && gap > GAP_MAX});
|
|
357
357
|
}
|
|
358
358
|
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
const outKeys = outWords.map((w) => key(w.word)).filter(Boolean);
|
|
366
|
-
|
|
367
|
-
const N = 4;
|
|
368
|
-
|
|
369
|
-
/**
|
|
370
|
-
* How often the SCRIPT says each phrase.
|
|
371
|
-
*
|
|
372
|
-
* "go to the code tab" appears twice in this script on purpose, and so does "I like how the".
|
|
373
|
-
* Flagging every repeated phrase would report both as stutters. A repeat is only a fault when
|
|
374
|
-
* the cut says something more often than the script does.
|
|
375
|
-
*/
|
|
376
|
-
const scriptGrams = new Map();
|
|
377
|
-
{
|
|
378
|
-
const sk = atoms.flatMap((a) => keysOf(a));
|
|
379
|
-
for (let i = 0; i + N <= sk.length; i++) {
|
|
380
|
-
const gram = sk.slice(i, i + N).join(' ');
|
|
381
|
-
scriptGrams.set(gram, (scriptGrams.get(gram) ?? 0) + 1);
|
|
382
|
-
}
|
|
383
|
-
}
|
|
359
|
+
const dropped = [...drop].sort((a, b) => a - b).map((i) => ({
|
|
360
|
+
from: Number(units[i].from.toFixed(2)),
|
|
361
|
+
to: Number(units[i].to.toFixed(2)),
|
|
362
|
+
reason: abandoned.has(i) ? 'announced-retake' : 'retake',
|
|
363
|
+
text: units[i].text.slice(0, 110),
|
|
364
|
+
}));
|
|
384
365
|
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
366
|
+
/* ----------------------------------------------------------------- verification */
|
|
367
|
+
|
|
368
|
+
const survivors = kept;
|
|
369
|
+
const stillAlike = [];
|
|
370
|
+
for (let i = 0; i < survivors.length; i++) {
|
|
371
|
+
for (let j = i + 1; j < survivors.length; j++) {
|
|
372
|
+
if (survivors[j].from - survivors[i].to > WINDOW) break;
|
|
373
|
+
if (similarity(survivors[i], survivors[j]) >= SIMILAR) {
|
|
374
|
+
stillAlike.push({
|
|
375
|
+
a: Number(survivors[i].from.toFixed(2)),
|
|
376
|
+
b: Number(survivors[j].from.toFixed(2)),
|
|
377
|
+
text: survivors[j].text.slice(0, 70),
|
|
378
|
+
});
|
|
379
|
+
}
|
|
393
380
|
}
|
|
394
|
-
seen.set(gram, [...previous, at]);
|
|
395
381
|
}
|
|
396
382
|
|
|
397
|
-
const
|
|
398
|
-
|
|
399
|
-
at: Number(w.start.toFixed(2)),
|
|
400
|
-
}));
|
|
401
|
-
|
|
383
|
+
const keptSeconds = keep.reduce((a, k) => a + (k.to - k.from), 0);
|
|
384
|
+
const sourceSeconds = words[words.length - 1].end - words[0].start;
|
|
402
385
|
const checks = {
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
keptSeconds: Number(
|
|
409
|
-
sourceSeconds: Number(
|
|
386
|
+
noRepeatsLeft: stillAlike.length === 0,
|
|
387
|
+
units: units.length,
|
|
388
|
+
retakeGroups: groups.length,
|
|
389
|
+
unitsDropped: drop.size,
|
|
390
|
+
spans: keep.length,
|
|
391
|
+
keptSeconds: Number(keptSeconds.toFixed(2)),
|
|
392
|
+
sourceSeconds: Number(sourceSeconds.toFixed(2)),
|
|
393
|
+
energyScored: medianRms != null,
|
|
410
394
|
};
|
|
411
395
|
|
|
412
396
|
/* ---------------------------------------------------------------------- output */
|
|
413
397
|
|
|
414
|
-
const stem = basename(wordsPath).replace(
|
|
398
|
+
const stem = basename(wordsPath).replace(/\.?words\.json$|\.json$/, '') || 'take';
|
|
415
399
|
const project = flag('project', null);
|
|
416
|
-
const outPath =
|
|
417
|
-
flag('out', null) ??
|
|
418
|
-
(project ? join('work', project, `${stem}.takes.json`) : join(dirname(wordsPath), `${stem}.takes.json`));
|
|
400
|
+
const outPath = flag('out', null) ?? (project ? join('work', project, `${stem}.takes.json`) : join(dirname(wordsPath), `${stem}.takes.json`));
|
|
419
401
|
mkdirSync(dirname(outPath), {recursive: true});
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
missing,
|
|
432
|
-
atoms: atoms.map((text, i) => ({
|
|
433
|
-
text,
|
|
434
|
-
chosen: chosen[i] ? {from: chosen[i].from, to: chosen[i].to, coverage: Number(chosen[i].coverage.toFixed(2))} : null,
|
|
435
|
-
attempts: takesPerAtom[i].length,
|
|
436
|
-
})),
|
|
437
|
-
},
|
|
438
|
-
null,
|
|
439
|
-
2,
|
|
440
|
-
),
|
|
441
|
-
);
|
|
442
|
-
|
|
443
|
-
const pct = ((checks.keptSeconds / checks.sourceSeconds) * 100).toFixed(0);
|
|
444
|
-
console.log(`${outPath}`);
|
|
445
|
-
console.log(` ${checks.placed}/${checks.atoms} script sentences placed`);
|
|
446
|
-
console.log(` ${checks.keptSeconds}s kept of ${checks.sourceSeconds}s (${pct}%)`);
|
|
447
|
-
console.log(` ${dropped.filter((d) => d.reason === 'retake').length} retakes, ${dropped.filter((d) => d.reason === 'off-script').length} off-script, ${dropped.filter((d) => d.reason === 'filler').length} filler dropped`);
|
|
448
|
-
|
|
449
|
-
let failed = false;
|
|
450
|
-
if (!checks.everyAtomPlaced) {
|
|
451
|
-
failed = true;
|
|
452
|
-
console.error(`\n${missing.length} script sentence(s) have no usable take:`);
|
|
453
|
-
for (const m of missing.slice(0, 10)) console.error(` - "${m.text.slice(0, 70)}" (${m.attempts} attempts found)`);
|
|
454
|
-
console.error(' Re-record these lines, or re-run with --loose to lower the match threshold.');
|
|
402
|
+
writeFileSync(outPath, JSON.stringify({fps: FPS, gapMax: GAP_MAX, gapKeep: GAP_KEEP, pad: PAD, checks, keep, gaps, groups: groupReport, dropped, scriptMissing}, null, 2));
|
|
403
|
+
|
|
404
|
+
console.log(outPath);
|
|
405
|
+
console.log(` ${checks.units} units, ${checks.retakeGroups} retake group(s), ${checks.unitsDropped} dropped`);
|
|
406
|
+
console.log(` ${checks.keptSeconds}s kept of ${checks.sourceSeconds}s (${((keptSeconds / sourceSeconds) * 100).toFixed(0)}%) in ${checks.spans} span(s)`);
|
|
407
|
+
console.log(` energy scoring: ${checks.energyScored ? 'on' : 'off (pass --media to enable)'}`);
|
|
408
|
+
if (groupReport.length) {
|
|
409
|
+
console.log('\n retakes dropped:');
|
|
410
|
+
for (const g of groupReport.slice(0, 8)) {
|
|
411
|
+
for (const d of g.dropped) console.log(` ${String(d.at).padStart(7)}s "${d.text}"`);
|
|
412
|
+
}
|
|
455
413
|
}
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
414
|
+
const beats = gaps.filter((g) => g.keptAsBeat).length;
|
|
415
|
+
if (beats) console.log(`\n ${beats} long pause(s) kept as deliberate beats before a short line.`);
|
|
416
|
+
if (scriptPath) {
|
|
417
|
+
console.log(`\n script check: ${scriptMissing.length} line(s) never said`);
|
|
418
|
+
for (const m of scriptMissing.slice(0, 6)) console.log(` - "${m.text.slice(0, 68)}"`);
|
|
460
419
|
}
|
|
461
|
-
if (
|
|
462
|
-
console.
|
|
420
|
+
if (stillAlike.length) {
|
|
421
|
+
console.error(`\n${stillAlike.length} near-duplicate(s) still in the cut:`);
|
|
422
|
+
for (const s of stillAlike.slice(0, 8)) console.error(` ${s.a}s and ${s.b}s — "${s.text}"`);
|
|
423
|
+
process.exit(1);
|
|
463
424
|
}
|
|
464
|
-
|
|
465
|
-
if (failed && !has('loose')) process.exit(1);
|