creator-editing-studio 1.2.1 → 1.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "creator-editing-studio",
3
- "version": "1.2.1",
3
+ "version": "1.2.2",
4
4
  "description": "Scaffolds a Creator Editing OS studio — components, scripts and rules for editing video from a plain-English brief.",
5
5
  "license": "UNLICENSED",
6
6
  "type": "module",
@@ -1,35 +1,38 @@
1
1
  #!/usr/bin/env node
2
2
  /**
3
- * Best-take selection: turn a script plus a messy take into a clean cut plan.
3
+ * Best-take selection: one messy take in, one clean cut plan out.
4
4
  *
5
- * npm run takes -- <script.md|txt> <words.json> [--project <name>] [--out <file>]
6
- * [--fps 30] [--gap-keep 0.18] [--min-coverage 0.6] [--loose]
5
+ * npm run takes -- <words.json> [--media <video>] [--script <script.md>]
6
+ * [--project <name>] [--out <file>] [--fps 30]
7
+ * [--similar 0.72] [--window 90] [--gap-max 0.6] [--gap-keep 0.25]
7
8
  *
8
- * THE PROBLEM THIS SOLVES. A talking head is read from a script, so the speaker fumbles and
9
- * says a sentence again. `trim.mjs` removes silence and filler, but it keeps every attempt —
10
- * so the cut contains the stumble AND the fix, and someone watches the whole thing to find
11
- * them. On the setup video that meant 16 drop ranges chosen by hand, several missed, and a
12
- * trip to CapCut to take out repeats nobody caught.
9
+ * WHAT IT IS FOR. A talking head is full of second attempts. Somebody fumbles, stops, says the
10
+ * sentence again, and both versions end up in the cut — so a human watches the whole thing to
11
+ * find them. This finds them instead, keeps the better one, and proves afterwards that none are
12
+ * left.
13
13
  *
14
- * WHY NOT JUST HUNT FOR REPEATS. Repeat-hunting is a filter: it finds duplicates it recognises
15
- * and misses the rest. Anchoring to the script is a constructor: each sentence of the script is
16
- * an atom, every atom gets exactly one span of audio, and anything that is not the winning span
17
- * for some atom never enters the cut. Repeats are not removed — they are impossible. Filler,
18
- * half-sentences, asides and dead air are gone for the same reason, without a rule for each.
14
+ * THE SPEECH IS THE TRUTH, NOT THE SCRIPT. An earlier version of this file anchored to the
15
+ * written script: every script sentence got one span and anything unmatched was dropped. It
16
+ * threw away six passages of real material on the first take it was pointed at — lines the
17
+ * speaker improvised on the day and meant to keep, including "you don't have to put the brief
18
+ * into the brief file". A repeat that survives is visible and fixable; a good line silently
19
+ * deleted is neither. So the default here is inverted:
19
20
  *
20
- * It also fails loudly. If a script sentence has no usable take, the build stops and names the
21
- * line, instead of handing over a cut with a hole for someone to find on playback.
21
+ * KEEP EVERYTHING, UNLESS IT IS A SECOND ATTEMPT AT SOMETHING ELSE THAT IS KEPT.
22
22
  *
23
- * WHAT IT WRITES work/<project>/<stem>.takes.json
24
- * keep[] {from, to, atom} source seconds to use, in order
25
- * dropped[] {from, to, reason, text}
26
- * atoms[] every script sentence, its chosen take and the attempts that lost
27
- * checks{} the verification results, all of which must pass
23
+ * Nothing has to justify its existence. A script, if there is one, is only a checker: it reports
24
+ * lines that never got said and breaks ties between close takes. It can never remove anything,
25
+ * and most people filming to camera do not have one.
28
26
  *
29
- * The shape of `keep[]` matches `trim.mjs`, so the rest of the pipeline is unchanged.
27
+ * WHAT IT WRITES work/<project>/<stem>.takes.json
28
+ * keep[] {from, to, units} source seconds to use, in order
29
+ * groups[] every retake cluster, which attempt won and why the others lost
30
+ * dropped[] what went, with the reason
31
+ * checks{} verification, all of which must pass
30
32
  */
31
33
  import {existsSync, mkdirSync, readFileSync, writeFileSync} from 'node:fs';
32
34
  import {basename, dirname, join} from 'node:path';
35
+ import {decodePcm} from './lib/media.mjs';
33
36
 
34
37
  /* ------------------------------------------------------------------------ args */
35
38
 
@@ -38,292 +41,304 @@ const flag = (name, fallback) => {
38
41
  const i = args.indexOf(`--${name}`);
39
42
  return i === -1 ? fallback : args[i + 1];
40
43
  };
41
- const has = (name) => args.includes(`--${name}`);
42
44
  const positional = args.filter((a, i) => !a.startsWith('--') && !String(args[i - 1] ?? '').startsWith('--'));
43
45
 
44
- const [scriptPath, wordsPath] = positional;
45
- if (!scriptPath || !wordsPath) {
46
- console.error('usage: npm run takes -- <script.md|txt> <words.json> [--project <name>]');
46
+ const wordsPath = positional[0];
47
+ if (!wordsPath) {
48
+ console.error('usage: npm run takes -- <words.json> [--media <video>] [--script <script.md>]');
47
49
  process.exit(1);
48
50
  }
49
51
 
50
52
  const FPS = Number(flag('fps', 30));
51
- /** A gap longer than this between kept spans is collapsed down to it. */
52
- const GAP_KEEP = Number(flag('gap-keep', 0.18));
53
- /** Below this share of an atom's words matched, a span is not a take of that atom at all. */
54
- const MIN_COVERAGE = Number(flag('min-coverage', 0.6));
55
- /** Padding kept around a span so cuts do not clip the first and last phoneme. */
56
- const PAD = 0.06;
53
+ /** How alike two passages must be before they count as attempts at the same thing. */
54
+ const SIMILAR = Number(flag('similar', 0.72));
55
+ /** Retakes happen close together. Two matching sentences minutes apart are both wanted. */
56
+ const WINDOW = Number(flag('window', 90));
57
+ /** A silence longer than this is dead air and gets collapsed. */
58
+ const GAP_MAX = Number(flag('gap-max', 0.6));
59
+ /** ...down to this. Not zero: speech with every gap removed sounds like a machine. */
60
+ const GAP_KEEP = Number(flag('gap-keep', 0.25));
61
+ /**
62
+ * Breathing room kept either side of a kept passage.
63
+ *
64
+ * 150ms, not the 60 an earlier version used. Below about 120 the cut lands close enough to the
65
+ * first phoneme to clip it, and the join reads as abrupt even when every word is intact.
66
+ */
67
+ const PAD = 0.15;
57
68
 
58
- /* ------------------------------------------------------------- script and words */
69
+ /* ------------------------------------------------------------------- the words */
70
+
71
+ const key = (t) => t.toLowerCase().replace(/['‘’]/g, '').replace(/[^a-z0-9]/g, '');
72
+
73
+ const words = JSON.parse(readFileSync(wordsPath, 'utf8')).words.filter((w) => w.start != null && w.end != null);
74
+ if (words.length < 10) {
75
+ console.error('Not enough timed words to work with.');
76
+ process.exit(1);
77
+ }
59
78
 
60
79
  /**
61
- * The script as atoms.
80
+ * Cues that announce a retake out loud.
62
81
  *
63
- * A sentence is the right unit: small enough that one fumble does not cost a paragraph, big
64
- * enough to identify uniquely in the transcript. Markdown furniture and the BRIEF.md comment
65
- * blocks are stripped — they are instructions to the writer, never spoken.
82
+ * When someone says one of these, the passage BEFORE it was abandoned — they are telling you so.
83
+ * Cheaper and more certain than any similarity measure.
66
84
  */
67
- const atomise = (raw) =>
68
- raw
69
- // Everything above the first `---` is front matter or a note to whoever wrote the script.
70
- // It is never spoken, and counting it as missing dialogue buries the lines that really are.
71
- .split(/^---\s*$/m)
72
- .slice(1)
73
- .join('\n')
74
- .replace(/<!--[\s\S]*?-->/g, '')
75
- .replace(/^#.*$/gm, '')
76
- .replace(/^[-*|>].*$/gm, '')
77
- .replace(/`[^`]*`/g, ' ')
78
- .split(/(?<=[.?!])\s+/)
79
- .map((s) => s.replace(/\s+/g, ' ').trim())
80
- .filter((s) => s.split(/\s+/).filter(Boolean).length >= 3);
85
+ const RESTART_CUES = [
86
+ 'let me redo that', 'let me try that again', 'one more time', 'sorry let me',
87
+ 'can i do that again', 'take two', 'scratch that', 'let me say that again',
88
+ 'start again', 'i will say that again',
89
+ ];
90
+
91
+ /* --------------------------------------------------------------- units of speech
92
+ * A unit is one spoken thought: it ends where the speaker put a full stop, or where they left a
93
+ * gap long enough to be a boundary rather than a breath. Working in units rather than fixed word
94
+ * counts means a retake is compared against the whole thought it replaced. */
95
+
96
+ const units = [];
97
+ {
98
+ let start = 0;
99
+ for (let i = 0; i < words.length; i++) {
100
+ const w = words[i];
101
+ const next = words[i + 1];
102
+ const gapAfter = next ? next.start - w.end : Infinity;
103
+ const endsSentence = /[.?!]$/.test(w.word);
104
+ if (endsSentence || gapAfter > 0.7 || i === words.length - 1) {
105
+ const span = words.slice(start, i + 1);
106
+ if (span.length) {
107
+ units.push({
108
+ index: units.length,
109
+ startIndex: start,
110
+ endIndex: i,
111
+ from: span[0].start,
112
+ to: span[span.length - 1].end,
113
+ text: span.map((x) => x.word).join(' '),
114
+ keys: span.map((x) => key(x.word)).filter(Boolean),
115
+ confidence: span.reduce((a, x) => a + (x.score ?? 0.8), 0) / span.length,
116
+ complete: endsSentence,
117
+ worstGap: span.slice(1).reduce((m, x, k) => Math.max(m, x.start - span[k].end), 0),
118
+ });
119
+ }
120
+ start = i + 1;
121
+ }
122
+ }
123
+ }
124
+
125
+ /* -------------------------------------------------------------- energy and pace */
81
126
 
82
- /** Comparable key for a word. Mirrors align_script.py so both agree on what "same" means. */
83
127
  /**
84
- * Spellings that mean the same word.
128
+ * Loudness per unit, so a flat read can lose to a committed one.
85
129
  *
86
- * The script is written in British English and the recogniser answers in American. Without this,
87
- * "licence" never matches "license" and a correctly spoken sentence is reported as missing while
88
- * the audio for it is thrown away as off-script. That is a worse failure than a repeat.
130
+ * Optional: without --media there is no audio to measure and the score simply does without it,
131
+ * rather than the tool refusing to run.
89
132
  */
90
- const SPELLING = new Map(Object.entries({
91
- license: 'licence', color: 'colour', colors: 'colours', favorite: 'favourite',
92
- realize: 'realise', organize: 'organise', gray: 'grey', analyze: 'analyse',
93
- }));
133
+ let rmsOf = () => null;
134
+ const mediaPath = flag('media', null);
135
+ if (mediaPath && existsSync(mediaPath)) {
136
+ const RATE = 22050;
137
+ const pcm = decodePcm(mediaPath, RATE);
138
+ if (pcm) {
139
+ rmsOf = (from, to) => {
140
+ const a = Math.max(0, Math.floor(from * RATE));
141
+ const b = Math.min(pcm.length, Math.ceil(to * RATE));
142
+ if (b <= a) return null;
143
+ let sum = 0;
144
+ for (let i = a; i < b; i++) sum += pcm[i] * pcm[i];
145
+ return Math.sqrt(sum / (b - a));
146
+ };
147
+ }
148
+ }
149
+
150
+ for (const u of units) {
151
+ u.seconds = u.to - u.from;
152
+ u.wordsPerSecond = u.keys.length / Math.max(0.3, u.seconds);
153
+ u.rms = rmsOf(u.from, u.to);
154
+ }
94
155
 
95
- const key = (token) => {
96
- const bare = token.toLowerCase().replace(/['‘’]/g, '').replace(/[^a-z0-9]/g, '');
97
- return SPELLING.get(bare) ?? bare;
156
+ const median = (xs) => {
157
+ const s = xs.filter((x) => x != null).sort((a, b) => a - b);
158
+ return s.length ? s[Math.floor(s.length / 2)] : null;
159
+ };
160
+ const medianRms = median(units.map((u) => u.rms));
161
+ const medianPace = median(units.map((u) => u.wordsPerSecond)) ?? 3;
162
+
163
+ /* ------------------------------------------------------------------ similarity */
164
+
165
+ /** Longest common subsequence length — tolerant of an inserted stumble or a swapped word. */
166
+ const lcs = (a, b) => {
167
+ const prev = new Array(b.length + 1).fill(0);
168
+ const cur = new Array(b.length + 1).fill(0);
169
+ for (let i = 1; i <= a.length; i++) {
170
+ for (let j = 1; j <= b.length; j++) {
171
+ cur[j] = a[i - 1] === b[j - 1] ? prev[j - 1] + 1 : Math.max(prev[j], cur[j - 1]);
172
+ }
173
+ for (let j = 0; j <= b.length; j++) prev[j] = cur[j];
174
+ }
175
+ return prev[b.length];
98
176
  };
99
177
 
100
178
  /**
101
- * Contractions, expanded.
179
+ * How alike two units are, measured against the SHORTER one.
102
180
  *
103
- * The script says "it'll" and the speaker says "it will" — or the other way round, depending on
104
- * the day and the recogniser. Both sides expand to the same pair of words so they compare equal.
105
- * Only forms carrying an apostrophe are touched, so "well" is never mistaken for "we will".
181
+ * Against the shorter, because an abandoned start — "so the main reason, so the main reason is
182
+ * that…" — is a short fragment entirely contained in the full attempt. Measured against the
183
+ * longer it would score low and be kept; against the shorter it scores 1.0 and is correctly
184
+ * recognised as the same thought, begun twice.
106
185
  */
107
- const CONTRACTION = new Map(Object.entries({
108
- itll: ['it', 'will'], theyll: ['they', 'will'], youll: ['you', 'will'], well: ['we', 'will'],
109
- itsa: ['it', 'is', 'a'], its: ['it', 'is'], thats: ['that', 'is'], whats: ['what', 'is'],
110
- theres: ['there', 'is'], heres: ['here', 'is'], lets: ['let', 'us'], youre: ['you', 'are'],
111
- theyre: ['they', 'are'], weve: ['we', 'have'], youve: ['you', 'have'], ive: ['i', 'have'],
112
- dont: ['do', 'not'], doesnt: ['does', 'not'], didnt: ['did', 'not'], isnt: ['is', 'not'],
113
- wasnt: ['was', 'not'], arent: ['are', 'not'], wont: ['will', 'not'], cant: ['can', 'not'],
114
- couldnt: ['could', 'not'], wouldnt: ['would', 'not'], shouldnt: ['should', 'not'],
115
- youd: ['you', 'would'], id: ['i', 'would'], wed: ['we', 'would'], ill: ['i', 'will'],
116
- }));
117
-
118
- /** One spoken or written word becomes the one or more keys it is equivalent to. */
119
- const keysOfWord = (token) => {
120
- const k = key(token);
121
- if (!k) return [];
122
- // Only expand when the original actually had an apostrophe, so ordinary words are left alone.
123
- if (/['‘’]/.test(token) && CONTRACTION.has(k)) return CONTRACTION.get(k);
124
- return [k];
186
+ const similarity = (a, b) => {
187
+ if (!a.keys.length || !b.keys.length) return 0;
188
+ const shorter = Math.min(a.keys.length, b.keys.length);
189
+ // Short units match each other on function words alone — "give it three things" against
190
+ // "this is the time you give it that" shares "give it" and scores 0.5 on a four-word unit.
191
+ // Below five words there is not enough signal to call anything a retake.
192
+ if (shorter < 5) return 0;
193
+ const overlap = lcs(a.keys, b.keys);
194
+ // And the overlap has to be real words, not two articles and a pronoun.
195
+ if (overlap < 4) return 0;
196
+ return overlap / shorter;
125
197
  };
126
198
 
127
- const keysOf = (text) => text.split(/[\s\-/]+/).flatMap(keysOfWord);
199
+ /* ---------------------------------------------------------------------- groups */
200
+
201
+ const groupOf = new Array(units.length).fill(-1);
202
+ const groups = [];
128
203
 
129
204
  /**
130
- * Filler. Deliberately short.
205
+ * Grouping is against a representative, never transitive.
131
206
  *
132
- * Stripping every "so" and "you know" makes speech sound sanded down, which reads as
133
- * over-edited and is its own kind of amateur. These are only ever used to PENALISE a take,
134
- * never to cut inside a winning one — if the best attempt at a line contains an "um", the line
135
- * keeps its "um" and sounds like a person.
207
+ * Chaining members together — A matches B, B matches C, so group them all — merges passages that
208
+ * have nothing to do with each other as soon as one ambiguous unit sits between them. It put
209
+ * "give it three things" in the same group as "they just won't really feel like you". A unit
210
+ * joins a group only if it is similar to that group's FIRST member, which keeps a group to one
211
+ * thought said more than once.
136
212
  */
137
- const FILLER = new Set(['um', 'uh', 'uhh', 'umm', 'erm', 'hmm', 'mmm', 'ah', 'eh']);
213
+ for (let i = 0; i < units.length; i++) {
214
+ if (groupOf[i] !== -1) continue;
215
+ const members = [i];
216
+ for (let j = i + 1; j < units.length; j++) {
217
+ if (groupOf[j] !== -1) continue;
218
+ if (units[j].from - units[i].to > WINDOW) break;
219
+ if (similarity(units[i], units[j]) < SIMILAR) continue;
220
+ members.push(j);
221
+ }
222
+ if (members.length < 2) continue;
223
+ groups.push({members});
224
+ for (const m of members) groupOf[m] = groups.length - 1;
225
+ }
138
226
 
139
- const words = JSON.parse(readFileSync(wordsPath, 'utf8')).words.filter((w) => w.start != null && w.end != null);
140
- const rawScript = readFileSync(scriptPath, 'utf8');
141
- // A plain .txt script has no `---` rule, so give it one — atomise() always drops what is above
142
- // the first rule, and without this a bare script would come back empty.
143
- const atoms = atomise(/^---\s*$/m.test(rawScript) ? rawScript : `---\n${rawScript}`);
144
- /** Each spoken word as the list of keys it is equivalent to. Index stays 1:1 with `words`. */
145
- const tokenKeys = words.map((w) => {
146
- const ks = keysOfWord(w.word);
147
- return ks.length ? ks : ['\u0000'];
148
- });
149
- const tokens = tokenKeys.map((ks) => ks[0]);
150
-
151
- if (!atoms.length) {
152
- console.error(`No script sentences found in ${scriptPath}.`);
153
- process.exit(1);
227
+ /** A spoken cue abandons whatever came immediately before it. */
228
+ const abandoned = new Set();
229
+ for (const u of units) {
230
+ const flat = u.keys.join(' ');
231
+ if (!RESTART_CUES.some((c) => flat.includes(c.replace(/[^a-z0-9 ]/g, '')))) continue;
232
+ const prev = units[u.index - 1];
233
+ if (prev) abandoned.add(prev.index);
154
234
  }
155
235
 
156
- /* --------------------------------------------------------------- finding takes */
236
+ /* --------------------------------------------------------------------- scoring */
157
237
 
158
238
  /**
159
- * Every span of transcript that is a plausible attempt at `atom`.
239
+ * One number per attempt.
160
240
  *
161
- * Greedy forward alignment from each position whose word matches the atom's first word, allowing
162
- * the speaker to skip, swap or insert a few words. A take is scored on how much of the sentence
163
- * actually made it out, how sure the aligner was, and whether it ran smoothly — a long pause or
164
- * an "um" inside a sentence is exactly what a fumbled attempt sounds like.
241
+ * Completeness dominates — a finished sentence beats a smoother fragment every time. Everything
242
+ * else nudges. The tie-break at the end is deliberate and matches how people actually record:
243
+ * they repeat a line until they are happy, so the last attempt is usually the keeper.
165
244
  */
166
- const findTakes = (atom, atomIndex) => {
167
- const want = keysOf(atom);
168
- if (!want.length) return [];
169
- const out = [];
170
- /** How far past the sentence the speaker may wander before the attempt is abandoned. */
171
- const window = Math.ceil(want.length * 2.2) + 6;
172
-
173
- /**
174
- * A take may begin on any of the sentence's first three words.
175
- *
176
- * Anchoring only on the first word loses a whole sentence to a single swapped article — the
177
- * script says "This system gets better", the speaker says "The system gets better", and the
178
- * line is reported missing while its audio is discarded. Coverage still has to clear the
179
- * threshold, so a take that drops its opening words scores lower rather than passing free.
180
- */
181
- const anchors = want.slice(0, 3);
182
-
183
- for (let start = 0; start < tokens.length; start++) {
184
- const anchorAt = anchors.indexOf(tokens[start]);
185
- if (anchorAt === -1) continue;
186
-
187
- let wi = anchorAt;
188
- let matched = anchorAt;
189
- let last = start;
190
- let extras = 0;
191
- for (let ti = start; ti < Math.min(tokens.length, start + window) && wi < want.length; ti++) {
192
- // A contraction covers more than one script word, so consume as many as it supplies.
193
- let consumed = 0;
194
- while (consumed < tokenKeys[ti].length && wi + consumed < want.length && tokenKeys[ti][consumed] === want[wi + consumed]) consumed++;
195
- if (consumed > 0) {
196
- matched += consumed;
197
- wi += consumed;
198
- last = ti;
199
- } else if (want.indexOf(tokens[ti], wi) !== -1 && want.indexOf(tokens[ti], wi) - wi <= 3) {
200
- // The speaker skipped a word or two and carried on; follow them.
201
- wi = want.indexOf(tokens[ti], wi) + 1;
202
- matched++;
203
- last = ti;
204
- } else {
205
- extras++;
206
- }
207
- }
208
-
209
- const coverage = matched / want.length;
210
- if (coverage < MIN_COVERAGE) continue;
211
-
212
- const from = words[start].start;
213
- const to = words[last].end;
214
- const span = words.slice(start, last + 1);
215
-
216
- let worstGap = 0;
217
- for (let i = 1; i < span.length; i++) worstGap = Math.max(worstGap, span[i].start - span[i - 1].end);
218
- const fillerInside = span.filter((w) => FILLER.has(key(w.word))).length;
219
- const confidence = span.reduce((a, w) => a + (w.score ?? 0.8), 0) / span.length;
220
-
221
- out.push({
222
- atomIndex,
223
- startIndex: start,
224
- endIndex: last,
225
- from,
226
- to,
227
- coverage,
228
- confidence,
229
- worstGap,
230
- fillerInside,
231
- extras,
232
- text: span.map((w) => w.word).join(' '),
233
- /**
234
- * One number, so the chooser has something to maximise.
235
- *
236
- * Coverage dominates: a complete sentence beats a smooth fragment every time. The small
237
- * bonus for starting later is there because when someone fumbles they fix it on the next
238
- * try, so with two otherwise equal attempts the second is the keeper.
239
- */
240
- score:
241
- coverage * 10 +
242
- confidence * 2 -
243
- Math.max(0, worstGap - 0.4) * 3 -
244
- fillerInside * 0.8 -
245
- (extras / want.length) * 1.5 +
246
- (start / tokens.length) * 0.3,
247
- });
248
-
249
- start = last; // attempts cannot overlap each other
250
- }
251
- return out;
245
+ const score = (u) => {
246
+ let s = 0;
247
+ s += u.complete ? 6 : 0;
248
+ s += Math.min(u.keys.length / 12, 1) * 3;
249
+ s += u.confidence * 2;
250
+ s -= Math.max(0, u.worstGap - 0.45) * 3;
251
+ if (abandoned.has(u.index)) s -= 20;
252
+ // Pace: too fast reads as rushed, too slow as laboured. Either direction costs the same.
253
+ s -= Math.min(2, Math.abs(u.wordsPerSecond - medianPace) / Math.max(0.8, medianPace) * 2);
254
+ // Energy, only when there was audio to measure.
255
+ if (u.rms != null && medianRms) s += Math.max(-1.5, Math.min(1.5, (u.rms / medianRms - 1) * 2));
256
+ return s;
252
257
  };
253
258
 
254
- const takesPerAtom = atoms.map((atom, i) => findTakes(atom, i));
259
+ for (const u of units) u.score = score(u);
255
260
 
256
- /* -------------------------------------------------------------- choosing takes */
261
+ /* ------------------------------------------------------- optional script checker
262
+ * A script never removes anything. It reports lines that were never said, and nudges a tie
263
+ * toward whichever attempt is closest to what was written. */
257
264
 
258
- /**
259
- * One take per atom, in script order, never overlapping.
260
- *
261
- * Greedy picking of each atom's best take breaks as soon as the best take of sentence 4 sits
262
- * earlier in the tape than the chosen take of sentence 5. This is weighted interval scheduling
263
- * over the atoms in order: for each candidate, the best total reachable while ending before the
264
- * next atom starts. Small enough to be exact rather than heuristic.
265
- */
266
- const NEG = -1e9;
267
- const best = takesPerAtom.map((takes) => takes.map(() => ({total: NEG, prev: -1})));
268
-
269
- for (let a = 0; a < atoms.length; a++) {
270
- const takes = takesPerAtom[a];
271
- for (let c = 0; c < takes.length; c++) {
272
- if (a === 0) {
273
- best[a][c] = {total: takes[c].score, prev: -1};
274
- continue;
275
- }
276
- let bestPrev = NEG;
277
- let bestPrevIndex = -1;
278
- for (let p = 0; p < takesPerAtom[a - 1].length; p++) {
279
- if (best[a - 1][p].total === NEG) continue;
280
- if (takesPerAtom[a - 1][p].endIndex >= takes[c].startIndex) continue;
281
- if (best[a - 1][p].total > bestPrev) {
282
- bestPrev = best[a - 1][p].total;
283
- bestPrevIndex = p;
284
- }
285
- }
286
- // An atom with no reachable predecessor can still start a run; the coverage check below
287
- // is what decides whether the result is acceptable, not this.
288
- best[a][c] =
289
- bestPrevIndex === -1
290
- ? {total: takes[c].score, prev: -1}
291
- : {total: bestPrev + takes[c].score, prev: bestPrevIndex};
292
- }
293
- }
265
+ const scriptPath = flag('script', null);
266
+ let scriptMissing = [];
267
+ if (scriptPath && existsSync(scriptPath)) {
268
+ const raw = readFileSync(scriptPath, 'utf8');
269
+ const body = /^---\s*$/m.test(raw) ? raw.split(/^---\s*$/m).slice(1).join('\n') : raw;
270
+ const lines = body
271
+ .replace(/<!--[\s\S]*?-->/g, '')
272
+ .replace(/^#.*$/gm, '')
273
+ .replace(/`[^`]*`/g, ' ')
274
+ .split(/(?<=[.?!])\s+/)
275
+ .map((s) => s.replace(/\s+/g, ' ').trim())
276
+ .filter((s) => s.split(/\s+/).length >= 4);
294
277
 
295
- const chosen = new Array(atoms.length).fill(null);
296
- let cursor = -1;
297
- for (let a = atoms.length - 1; a >= 0; a--) {
298
- const takes = takesPerAtom[a];
299
- if (!takes.length) continue;
300
- let pick = cursor;
301
- if (pick === -1) {
302
- let top = NEG;
303
- for (let c = 0; c < takes.length; c++)
304
- if (best[a][c].total > top) {
305
- top = best[a][c].total;
306
- pick = c;
307
- }
278
+ for (const line of lines) {
279
+ const lk = line.split(/\s+/).map(key).filter(Boolean);
280
+ const best = Math.max(0, ...units.map((u) => lcs(lk, u.keys) / Math.max(1, lk.length)));
281
+ if (best < 0.6) scriptMissing.push({text: line, bestMatch: Number(best.toFixed(2))});
282
+ }
283
+ // Closest-to-script wins a close call, but only by a little.
284
+ for (const g of groups) {
285
+ for (const m of g.members) {
286
+ const u = units[m];
287
+ const best = Math.max(0, ...lines.map((l) => lcs(l.split(/\s+/).map(key).filter(Boolean), u.keys) / Math.max(1, u.keys.length)));
288
+ u.score += best * 1.2;
289
+ }
308
290
  }
309
- if (pick === -1 || !takes[pick]) continue;
310
- chosen[a] = takes[pick];
311
- cursor = best[a][pick].prev;
312
291
  }
313
292
 
314
- /* ------------------------------------------------------------------ the result */
293
+ /* ----------------------------------------------------------------- the decision */
315
294
 
316
- const missing = [];
317
- atoms.forEach((atom, i) => {
318
- if (!chosen[i]) missing.push({index: i, text: atom, attempts: takesPerAtom[i].length});
319
- });
295
+ const drop = new Set();
296
+ const groupReport = [];
320
297
 
321
- const keep = chosen
322
- .filter(Boolean)
323
- .map((t) => ({from: Math.max(0, t.from - PAD), to: t.to + PAD, atom: t.atomIndex, text: t.text}))
324
- .sort((a, b) => a.from - b.from);
298
+ for (const g of groups) {
299
+ const members = [...new Set(g.members)].sort((a, b) => a - b);
300
+ let winner = members[0];
301
+ for (const m of members) {
302
+ const a = units[m];
303
+ const b = units[winner];
304
+ // Within a hair of each other, the later one wins.
305
+ if (a.score > b.score + 0.3 || (Math.abs(a.score - b.score) <= 0.3 && a.from > b.from)) winner = m;
306
+ }
307
+ for (const m of members) if (m !== winner) drop.add(m);
308
+ groupReport.push({
309
+ kept: {at: Number(units[winner].from.toFixed(2)), text: units[winner].text.slice(0, 90), score: Number(units[winner].score.toFixed(2))},
310
+ dropped: members.filter((m) => m !== winner).map((m) => ({
311
+ at: Number(units[m].from.toFixed(2)),
312
+ text: units[m].text.slice(0, 90),
313
+ score: Number(units[m].score.toFixed(2)),
314
+ })),
315
+ });
316
+ }
317
+ for (const i of abandoned) drop.add(i);
318
+
319
+ const kept = units.filter((u) => !drop.has(u.index));
320
+
321
+ /* ------------------------------------------------------------- building the plan
322
+ * Consecutive survivors become one span, so the cut is not chopped at every sentence. A gap
323
+ * between spans is collapsed only when it is longer than a breath — and a pause in front of a
324
+ * short line is usually deliberate, so those are left alone. */
325
+
326
+ const keep = [];
327
+ for (const u of kept) {
328
+ const last = keep[keep.length - 1];
329
+ if (last && u.startIndex === last.endIndex + 1 && u.from - last.to < GAP_MAX) {
330
+ last.to = u.to;
331
+ last.endIndex = u.endIndex;
332
+ last.units.push(u.index);
333
+ } else {
334
+ keep.push({from: u.from, to: u.to, startIndex: u.startIndex, endIndex: u.endIndex, units: [u.index]});
335
+ }
336
+ }
325
337
 
326
- // Overlapping padding between neighbours would double a word; meet in the middle instead.
338
+ for (const k of keep) {
339
+ k.from = Math.max(0, k.from - PAD);
340
+ k.to += PAD;
341
+ }
327
342
  for (let i = 1; i < keep.length; i++) {
328
343
  if (keep[i].from < keep[i - 1].to) {
329
344
  const mid = (keep[i].from + keep[i - 1].to) / 2;
@@ -332,134 +347,78 @@ for (let i = 1; i < keep.length; i++) {
332
347
  }
333
348
  }
334
349
 
335
- const keptIndexes = new Set();
336
- chosen.filter(Boolean).forEach((t) => {
337
- for (let i = t.startIndex; i <= t.endIndex; i++) keptIndexes.add(i);
338
- });
339
-
340
- const dropped = [];
341
- let runStart = null;
342
- for (let i = 0; i < words.length; i++) {
343
- const inCut = keptIndexes.has(i);
344
- if (!inCut && runStart === null) runStart = i;
345
- if ((inCut || i === words.length - 1) && runStart !== null) {
346
- const end = inCut ? i - 1 : i;
347
- const text = words.slice(runStart, end + 1).map((w) => w.word).join(' ');
348
- const alsoKept = chosen.filter(Boolean).some((t) => keysOf(t.text).join(' ').includes(keysOf(text).join(' ')));
349
- dropped.push({
350
- from: words[runStart].start,
351
- to: words[end].end,
352
- reason: alsoKept ? 'retake' : text.split(/\s+/).length <= 2 ? 'filler' : 'off-script',
353
- text,
354
- });
355
- runStart = null;
356
- }
350
+ /** Gaps between spans, after the cut: long dead air collapsed, deliberate beats kept. */
351
+ const gaps = [];
352
+ for (let i = 1; i < keep.length; i++) {
353
+ const gap = keep[i].from - keep[i - 1].to;
354
+ const nextIsShort = units[keep[i].units[0]].keys.length <= 6;
355
+ const target = gap > GAP_MAX && !nextIsShort ? GAP_KEEP : gap;
356
+ gaps.push({after: i - 1, was: Number(gap.toFixed(2)), becomes: Number(target.toFixed(2)), keptAsBeat: target === gap && gap > GAP_MAX});
357
357
  }
358
358
 
359
- /* ----------------------------------------------------------------- verification
360
- * Run on the transcript the cut WILL have, worked out from the spans kept, so there is no
361
- * second transcription and no waiting. Every one of these must pass; a build that fails here
362
- * is a build that would have sent someone to CapCut. */
363
-
364
- const outWords = [...keptIndexes].sort((a, b) => a - b).map((i) => words[i]);
365
- const outKeys = outWords.map((w) => key(w.word)).filter(Boolean);
366
-
367
- const N = 4;
368
-
369
- /**
370
- * How often the SCRIPT says each phrase.
371
- *
372
- * "go to the code tab" appears twice in this script on purpose, and so does "I like how the".
373
- * Flagging every repeated phrase would report both as stutters. A repeat is only a fault when
374
- * the cut says something more often than the script does.
375
- */
376
- const scriptGrams = new Map();
377
- {
378
- const sk = atoms.flatMap((a) => keysOf(a));
379
- for (let i = 0; i + N <= sk.length; i++) {
380
- const gram = sk.slice(i, i + N).join(' ');
381
- scriptGrams.set(gram, (scriptGrams.get(gram) ?? 0) + 1);
382
- }
383
- }
359
+ const dropped = [...drop].sort((a, b) => a - b).map((i) => ({
360
+ from: Number(units[i].from.toFixed(2)),
361
+ to: Number(units[i].to.toFixed(2)),
362
+ reason: abandoned.has(i) ? 'announced-retake' : 'retake',
363
+ text: units[i].text.slice(0, 110),
364
+ }));
384
365
 
385
- const repeats = [];
386
- const seen = new Map();
387
- for (let i = 0; i + N <= outKeys.length; i++) {
388
- const gram = outKeys.slice(i, i + N).join(' ');
389
- const at = outWords[i].start;
390
- const previous = seen.get(gram) ?? [];
391
- if (previous.length >= (scriptGrams.get(gram) ?? 0) && previous.some((t) => at - t < 20)) {
392
- repeats.push({gram, first: Number(previous[previous.length - 1].toFixed(2)), again: Number(at.toFixed(2))});
366
+ /* ----------------------------------------------------------------- verification */
367
+
368
+ const survivors = kept;
369
+ const stillAlike = [];
370
+ for (let i = 0; i < survivors.length; i++) {
371
+ for (let j = i + 1; j < survivors.length; j++) {
372
+ if (survivors[j].from - survivors[i].to > WINDOW) break;
373
+ if (similarity(survivors[i], survivors[j]) >= SIMILAR) {
374
+ stillAlike.push({
375
+ a: Number(survivors[i].from.toFixed(2)),
376
+ b: Number(survivors[j].from.toFixed(2)),
377
+ text: survivors[j].text.slice(0, 70),
378
+ });
379
+ }
393
380
  }
394
- seen.set(gram, [...previous, at]);
395
381
  }
396
382
 
397
- const survivingFiller = outWords.filter((w) => FILLER.has(key(w.word))).map((w) => ({
398
- word: w.word,
399
- at: Number(w.start.toFixed(2)),
400
- }));
401
-
383
+ const keptSeconds = keep.reduce((a, k) => a + (k.to - k.from), 0);
384
+ const sourceSeconds = words[words.length - 1].end - words[0].start;
402
385
  const checks = {
403
- everyAtomPlaced: missing.length === 0,
404
- noRepeatedPhrase: repeats.length === 0,
405
- fillerLeftInside: survivingFiller.length,
406
- atoms: atoms.length,
407
- placed: atoms.length - missing.length,
408
- keptSeconds: Number(keep.reduce((a, k) => a + (k.to - k.from), 0).toFixed(2)),
409
- sourceSeconds: Number((words[words.length - 1].end - words[0].start).toFixed(2)),
386
+ noRepeatsLeft: stillAlike.length === 0,
387
+ units: units.length,
388
+ retakeGroups: groups.length,
389
+ unitsDropped: drop.size,
390
+ spans: keep.length,
391
+ keptSeconds: Number(keptSeconds.toFixed(2)),
392
+ sourceSeconds: Number(sourceSeconds.toFixed(2)),
393
+ energyScored: medianRms != null,
410
394
  };
411
395
 
412
396
  /* ---------------------------------------------------------------------- output */
413
397
 
414
- const stem = basename(wordsPath).replace(/\.words\.json$|\.json$/, '');
398
+ const stem = basename(wordsPath).replace(/\.?words\.json$|\.json$/, '') || 'take';
415
399
  const project = flag('project', null);
416
- const outPath =
417
- flag('out', null) ??
418
- (project ? join('work', project, `${stem}.takes.json`) : join(dirname(wordsPath), `${stem}.takes.json`));
400
+ const outPath = flag('out', null) ?? (project ? join('work', project, `${stem}.takes.json`) : join(dirname(wordsPath), `${stem}.takes.json`));
419
401
  mkdirSync(dirname(outPath), {recursive: true});
420
-
421
- writeFileSync(
422
- outPath,
423
- JSON.stringify(
424
- {
425
- fps: FPS,
426
- gapKeep: GAP_KEEP,
427
- checks,
428
- keep,
429
- dropped,
430
- repeats,
431
- missing,
432
- atoms: atoms.map((text, i) => ({
433
- text,
434
- chosen: chosen[i] ? {from: chosen[i].from, to: chosen[i].to, coverage: Number(chosen[i].coverage.toFixed(2))} : null,
435
- attempts: takesPerAtom[i].length,
436
- })),
437
- },
438
- null,
439
- 2,
440
- ),
441
- );
442
-
443
- const pct = ((checks.keptSeconds / checks.sourceSeconds) * 100).toFixed(0);
444
- console.log(`${outPath}`);
445
- console.log(` ${checks.placed}/${checks.atoms} script sentences placed`);
446
- console.log(` ${checks.keptSeconds}s kept of ${checks.sourceSeconds}s (${pct}%)`);
447
- console.log(` ${dropped.filter((d) => d.reason === 'retake').length} retakes, ${dropped.filter((d) => d.reason === 'off-script').length} off-script, ${dropped.filter((d) => d.reason === 'filler').length} filler dropped`);
448
-
449
- let failed = false;
450
- if (!checks.everyAtomPlaced) {
451
- failed = true;
452
- console.error(`\n${missing.length} script sentence(s) have no usable take:`);
453
- for (const m of missing.slice(0, 10)) console.error(` - "${m.text.slice(0, 70)}" (${m.attempts} attempts found)`);
454
- console.error(' Re-record these lines, or re-run with --loose to lower the match threshold.');
402
+ writeFileSync(outPath, JSON.stringify({fps: FPS, gapMax: GAP_MAX, gapKeep: GAP_KEEP, pad: PAD, checks, keep, gaps, groups: groupReport, dropped, scriptMissing}, null, 2));
403
+
404
+ console.log(outPath);
405
+ console.log(` ${checks.units} units, ${checks.retakeGroups} retake group(s), ${checks.unitsDropped} dropped`);
406
+ console.log(` ${checks.keptSeconds}s kept of ${checks.sourceSeconds}s (${((keptSeconds / sourceSeconds) * 100).toFixed(0)}%) in ${checks.spans} span(s)`);
407
+ console.log(` energy scoring: ${checks.energyScored ? 'on' : 'off (pass --media to enable)'}`);
408
+ if (groupReport.length) {
409
+ console.log('\n retakes dropped:');
410
+ for (const g of groupReport.slice(0, 8)) {
411
+ for (const d of g.dropped) console.log(` ${String(d.at).padStart(7)}s "${d.text}"`);
412
+ }
455
413
  }
456
- if (repeats.length) {
457
- failed = true;
458
- console.error(`\n${repeats.length} phrase(s) still repeat in the cut:`);
459
- for (const r of repeats.slice(0, 10)) console.error(` - "${r.gram}" at ${r.first}s and again at ${r.again}s`);
414
+ const beats = gaps.filter((g) => g.keptAsBeat).length;
415
+ if (beats) console.log(`\n ${beats} long pause(s) kept as deliberate beats before a short line.`);
416
+ if (scriptPath) {
417
+ console.log(`\n script check: ${scriptMissing.length} line(s) never said`);
418
+ for (const m of scriptMissing.slice(0, 6)) console.log(` - "${m.text.slice(0, 68)}"`);
460
419
  }
461
- if (survivingFiller.length) {
462
- console.log(`\n ${survivingFiller.length} filler word(s) kept inside winning takes — left in on purpose.`);
420
+ if (stillAlike.length) {
421
+ console.error(`\n${stillAlike.length} near-duplicate(s) still in the cut:`);
422
+ for (const s of stillAlike.slice(0, 8)) console.error(` ${s.a}s and ${s.b}s — "${s.text}"`);
423
+ process.exit(1);
463
424
  }
464
-
465
- if (failed && !has('loose')) process.exit(1);