@ossclip/core 0.1.6 → 0.1.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ossclip/core",
3
- "version": "0.1.6",
3
+ "version": "0.1.9",
4
4
  "description": "ossclip's framework-free pipeline: schema, transcription, analysis, cutlist, captions, framing, and the LLM producer",
5
5
  "type": "module",
6
6
  "license": "MIT",
package/src/blooper.ts CHANGED
@@ -1,5 +1,6 @@
1
1
  import { normalizeToken } from "./analyze";
2
2
  import { isSentenceStart } from "./clip";
3
+ import { levenshtein } from "./phonetics";
3
4
  import type { Transcript } from "./schema";
4
5
 
5
6
  /**
@@ -14,8 +15,14 @@ import type { Transcript } from "./schema";
14
15
  *
15
16
  * A marker the speaker says OUT LOUD is the deterministic subset. It needs no
16
17
  * judgement: the word is in the transcript or it is not. So this ships the
17
- * useful half of the feature and leaves the guarantee intact — the semantic
18
- * detector remains unbuilt, deliberately.
18
+ * useful half of the feature and leaves the guarantee intact.
19
+ *
20
+ * The OTHER half — the flub the speaker did NOT mark — turned out to have a
21
+ * deterministic formulation too: `retake.ts` (R27 §128) collapses consecutive
22
+ * near-identical sentences by token similarity, no LLM, same purity
23
+ * guarantee. What's left unbuilt is narrower than this comment used to claim:
24
+ * a genuinely REWORDED retake (different words, same idea) is still semantic,
25
+ * and stays out of scope on purpose (ROADMAP.md).
19
26
  *
20
27
  * The pattern, from the take that motivated it:
21
28
  *
@@ -38,6 +45,57 @@ export interface BloopSpan {
38
45
  endSec: number;
39
46
  /** How many marker words this span swallowed — 2+ means repeated attempts. */
40
47
  markers: number;
48
+ /** The marker text this span was searched for, normalized — for the report line. */
49
+ marker: string;
50
+ /**
51
+ * Surface forms in this span that matched by sound-alike or edit distance,
52
+ * not exact text — e.g. ASR wrote "looker" for a "blooper" marker. Every
53
+ * fuzzy hit must land here: it is what makes fuzzy matching safe to ship
54
+ * on by default, since a false positive shows up in report.txt instead of
55
+ * silently cutting a good take (Task 3, editor-dogfood-fixes plan).
56
+ */
57
+ matched: string[];
58
+ }
59
+
60
+ // Fuzzy matching only turns on once the marker is long enough that a false
61
+ // positive is unlikely — a short marker like "cut" sound-alikes ("cat") and
62
+ // sits within edit distance 2 of half the dictionary ("but", "gut", "cot"),
63
+ // so short markers stay exact-only (Task 3, editor-dogfood-fixes plan).
64
+ const FUZZY_MIN_MARKER_LEN = 6;
65
+ // "blooper" → "looker" is exactly this: 2 edits (drop the "b", substitute
66
+ // "p" for "k"). Found in the wild — see the guard test in blooper.test.ts.
67
+ const FUZZY_MAX_DISTANCE = 2;
68
+
69
+ interface MarkerMatch {
70
+ /** Normalized text of the transcript word that matched. */
71
+ surface: string;
72
+ /** False when this needed sound-alike/edit-distance rather than an exact hit. */
73
+ exact: boolean;
74
+ }
75
+
76
+ /**
77
+ * Whether a transcript word counts as the marker, and how.
78
+ *
79
+ * Exact match (today's rule) always wins first. Past that, the only fuzzy
80
+ * arm is a small edit distance — NOT `soundsSimilar` (§125,
81
+ * PHASE1-FINDINGS.md). The first field run of this feature paired the two
82
+ * arms as designed and got the worst of both: `soundsSimilar("builds",
83
+ * "blooper")` is true (shared "b" onset, score over its 0.34 floor) and cut
84
+ * 86.8% of a 125.9s video, while the pair `soundsSimilar` exists to catch —
85
+ * "looker" for "blooper" — is REJECTED by its own onset test (b/l differ)
86
+ * and only ever matched via Levenshtein anyway. Sound-alike was admitting
87
+ * garbage and catching nothing real, so it is gone; Levenshtein alone still
88
+ * catches "looker" (distance 2) and does not catch "builds" (distance 6).
89
+ */
90
+ function matchMarker(wordText: string, want: string): MarkerMatch | null {
91
+ const norm = normalizeToken(wordText);
92
+ if (!norm) return null;
93
+ if (norm === want) return { surface: norm, exact: true };
94
+ if (want.length < FUZZY_MIN_MARKER_LEN) return null;
95
+ if (levenshtein(norm, want) <= FUZZY_MAX_DISTANCE) {
96
+ return { surface: norm, exact: false };
97
+ }
98
+ return null;
41
99
  }
42
100
 
43
101
  /**
@@ -45,7 +103,9 @@ export interface BloopSpan {
45
103
  *
46
104
  * `marker` is matched with `normalizeToken`, the same normalizer the filler
47
105
  * detector uses, so case and trailing punctuation do not matter — ASR writes
48
- * the word as "blooper." with the period riding on it.
106
+ * the word as "blooper." with the period riding on it. Beyond exact text, a
107
+ * marker of at least `FUZZY_MIN_MARKER_LEN` characters also matches an ASR
108
+ * mishearing — see `matchMarker`.
49
109
  *
50
110
  * Returns spans in transcript order, non-overlapping.
51
111
  */
@@ -53,11 +113,15 @@ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpa
53
113
  const want = normalizeToken(marker);
54
114
  if (!want) return [];
55
115
  const words = transcript.words;
56
- const isMarker = (i: number): boolean => normalizeToken(words[i]?.text ?? "") === want;
116
+ const matchAt = (i: number): MarkerMatch | null => {
117
+ const w = words[i];
118
+ return w ? matchMarker(w.text, want) : null;
119
+ };
57
120
 
58
121
  const spans: BloopSpan[] = [];
59
122
  for (let i = 0; i < words.length; i++) {
60
- if (!isMarker(i)) continue;
123
+ const match = matchAt(i);
124
+ if (!match) continue;
61
125
 
62
126
  // Walk back over the attempt this marker spoiled, to the start of its
63
127
  // sentence. The marker's own text usually ENDS a sentence ("blooper."), so
@@ -70,10 +134,12 @@ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpa
70
134
  // one-word island of a sentence nobody finished.
71
135
  const prev = spans[spans.length - 1];
72
136
  let markers = 1;
137
+ let matched = match.exact ? [] : [match.surface];
73
138
  if (prev && start <= prev.endWord + 1) {
74
139
  spans.pop();
75
140
  start = prev.startWord;
76
141
  markers = prev.markers + 1;
142
+ matched = [...prev.matched, ...matched];
77
143
  }
78
144
 
79
145
  spans.push({
@@ -82,6 +148,8 @@ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpa
82
148
  startSec: words[start]!.start,
83
149
  endSec: words[i]!.end,
84
150
  markers,
151
+ marker: want,
152
+ matched,
85
153
  });
86
154
  }
87
155
  return spans;
@@ -98,5 +166,13 @@ export function formatBloopSpan(transcript: Transcript, span: BloopSpan): string
98
166
  .map((w) => w.text)
99
167
  .join(" ");
100
168
  const attempts = span.markers > 1 ? ` (${span.markers} attempts)` : "";
101
- return `"${said}"${attempts}`;
169
+ // A fuzzy hit must never be silent — this line is the safety net that
170
+ // makes on-by-default fuzzy matching acceptable (Task 3, editor-dogfood-fixes
171
+ // plan): a false positive shows up here instead of quietly cutting a good
172
+ // take.
173
+ const fuzzy =
174
+ span.matched.length > 0
175
+ ? " " + span.matched.map((m) => `matched "${m}" ~ "${span.marker}"`).join(", ")
176
+ : "";
177
+ return `"${said}"${attempts}${fuzzy}`;
102
178
  }
package/src/concat.ts ADDED
@@ -0,0 +1,333 @@
1
+ import { existsSync } from "node:fs";
2
+ import { readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises";
3
+ import { join } from "node:path";
4
+ import { z } from "zod/v4";
5
+ import { probe, type IngestTools } from "./ingest";
6
+ import { run } from "./exec";
7
+
8
+ /**
9
+ * `ossclip produce <folder>` (2026-08-05 field request, verbatim intent in
10
+ * .superpowers/sdd/folder-input-brief.md): a folder of camera-clip takes gets
11
+ * concatenated into ONE source before the normal produce pipeline ever sees
12
+ * it, instead of the user hand-concatenating with an agent first.
13
+ *
14
+ * Split per CLAUDE.md's pure/IO mandate — the bug that motivates the split is
15
+ * in the brief: a hand-built ffmpeg filtergraph corrupted by shell expansion
16
+ * (zsh's `:a` history modifier) silently played clip 1's audio in every
17
+ * concat slot, and nothing validated the filter STRING before it reached
18
+ * ffmpeg. `planFolderConcat`, `buildConcatFilter`, `folderManifestKey` and
19
+ * `assertAllClipsHaveAudio` are pure so each can be asserted on directly;
20
+ * `listFolderVideos` and `concatFolder` are the only I/O.
21
+ */
22
+
23
+ const VIDEO_EXTENSIONS = ["mov", "mp4", "m4v", "mkv", "webm", "avi"] as const;
24
+ const VIDEO_EXTENSION_SET = new Set<string>(VIDEO_EXTENSIONS);
25
+
26
+ function noVideoFilesError(folder: string): Error {
27
+ return new Error(
28
+ `no video files found directly inside ${folder} ` +
29
+ `(looked for: ${VIDEO_EXTENSIONS.map((e) => `.${e}`).join(", ")})`,
30
+ );
31
+ }
32
+
33
+ export interface ConcatEntry {
34
+ name: string;
35
+ mtimeMs: number;
36
+ size: number;
37
+ }
38
+
39
+ /**
40
+ * Order clips for concatenation. `name` (default, per the field request "sort
41
+ * them by name or date modified, name being default") is a PLAIN codepoint
42
+ * sort — comparing strings with `<`/`>` rather than `localeCompare`, which is
43
+ * what `ls` gives on a case-sensitive filesystem and a locale-aware sort would
44
+ * NOT: it reorders case and punctuation differently per machine locale, which
45
+ * would make the same folder concat in a different order on a different
46
+ * machine. `mtime` ties (a batch copy that preserved one timestamp across
47
+ * several files) fall back to name so the order stays reproducible either way.
48
+ */
49
+ export function planFolderConcat(
50
+ entries: readonly ConcatEntry[],
51
+ sort: "name" | "mtime",
52
+ ): string[] {
53
+ const byName = (a: ConcatEntry, b: ConcatEntry): number =>
54
+ a.name < b.name ? -1 : a.name > b.name ? 1 : 0;
55
+ const cmp = sort === "name" ? byName : (a: ConcatEntry, b: ConcatEntry) => a.mtimeMs - b.mtimeMs || byName(a, b);
56
+ return [...entries].sort(cmp).map((e) => e.name);
57
+ }
58
+
59
+ /**
60
+ * A deterministic, order-independent identity for a folder's clip set —
61
+ * sorted by name (a canonical order regardless of `readdir`'s OS-dependent
62
+ * enumeration order) before joining, so the same files always hash the same
63
+ * way. `sort` is folded in because a `--sort` flip changes the concat's
64
+ * actual bytes (different clip order), not just how it was chosen.
65
+ *
66
+ * Fix for a review finding on the first cut of this feature: the workdir
67
+ * hash used to be derived from the FOLDER PATH alone, which is stable across
68
+ * content changes — but `audio.wav`, `transcript.json`, the content-rect
69
+ * cache and the mezzanine are all existence-keyed inside that same workdir.
70
+ * Adding a take (or flipping --sort) rebuilt `source-concat.mp4` correctly
71
+ * but silently reused every one of those, producing a video with captions
72
+ * transcribed against the PREVIOUS concat. Hashing the manifest content here
73
+ * — the same invariant a file input already has via `sha1File` — means a
74
+ * changed folder gets a fresh workdir, and every derived cache is fresh too.
75
+ *
76
+ * Serialized with JSON.stringify, not a `:`/`|` delimiter join (audit fix):
77
+ * a filename is user-controlled free text that can itself contain the
78
+ * delimiters, letting two DIFFERENT entry sets serialize to one identical
79
+ * key — `a:1` sized 2 and `a` sized `1:2` collide under a `:` join, and a
80
+ * collision here means one folder silently reuses another's transcript and
81
+ * mezzanine. JSON escapes the filename instead of trusting it. This changed
82
+ * every existing folder workdir hash once — a one-time cache invalidation
83
+ * (fresh workdir, full re-concat/re-transcribe on the next run), accepted as
84
+ * the cost of an injection-proof key.
85
+ */
86
+ export function folderManifestKey(entries: readonly ConcatEntry[], sort: "name" | "mtime"): string {
87
+ const canonical = [...entries]
88
+ .sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0))
89
+ .map((e) => ({ name: e.name, size: e.size, mtimeMs: e.mtimeMs }));
90
+ return JSON.stringify({ sort, entries: canonical });
91
+ }
92
+
93
+ /**
94
+ * The `-filter_complex` string concatenating `n` inputs into one output.
95
+ * Each input gets its OWN scale+pad+fps+setsar+format chain — letterboxed to
96
+ * `target`, never cropped, since produce's own framing decides crops later —
97
+ * and its own audio resample, before the `concat` filter joins them.
98
+ *
99
+ * Rotation: ffmpeg auto-rotates on decode by default (R27 §119, `probe()` in
100
+ * ingest.ts relies on the same fact), so every `[i:v]` input here is ASSUMED
101
+ * to already arrive in DISPLAYED orientation — the scale/pad math needs no
102
+ * separate rotation step. This is an assumption carried over from R27 §119,
103
+ * not something the folder-input verification run could independently
104
+ * confirm: a letterboxed, correctly-proportioned 1080x1920 output is also
105
+ * what a WRONG rotation assumption would produce once padded to a portrait
106
+ * canvas, so that run couldn't distinguish "handled correctly" from
107
+ * "accidentally looks fine." Flagged rather than overclaimed per CLAUDE.md.
108
+ *
109
+ * `n` labels of each kind, never more or fewer, and the tail's `[vI][aI]`
110
+ * pairs are built in the SAME loop that emits them — this is the direct
111
+ * regression test target for the field bug (see module comment): a
112
+ * hand-built graph had a slot silently reference input 0's audio a second
113
+ * time instead of its own index, and nothing caught the STRING being wrong.
114
+ */
115
+ export function buildConcatFilter(n: number, target: { w: number; h: number }): string {
116
+ const chains: string[] = [];
117
+ const tail: string[] = [];
118
+ for (let i = 0; i < n; i++) {
119
+ chains.push(
120
+ `[${i}:v]scale=${target.w}:${target.h}:force_original_aspect_ratio=decrease,` +
121
+ `pad=${target.w}:${target.h}:(ow-iw)/2:(oh-ih)/2,fps=30,setsar=1,format=yuv420p[v${i}]`,
122
+ );
123
+ chains.push(`[${i}:a]aresample=48000,aformat=channel_layouts=stereo[a${i}]`);
124
+ tail.push(`[v${i}][a${i}]`);
125
+ }
126
+ return `${chains.join(";")};${tail.join("")}concat=n=${n}:v=1:a=1[outv][outa]`;
127
+ }
128
+
129
+ /**
130
+ * `buildConcatFilter` emits `[i:a]` unconditionally for every input — it has
131
+ * no way to know a clip is silent. Handed a video-only clip (b-roll with no
132
+ * audio stream), ffmpeg dies deep inside the filtergraph with a bare stream-
133
+ * specifier error that names neither the clip nor the reason. `probe()`
134
+ * already reports `hasAudio`; failing HERE, before ffmpeg ever runs, keeps
135
+ * faith with the brief's "a file that probe() rejects is an error naming the
136
+ * file, not a silent skip" — a clip with no audio is the same class of
137
+ * problem, just discovered an instant later than "no video stream at all".
138
+ */
139
+ export function assertAllClipsHaveAudio(
140
+ clips: ReadonlyArray<{ name: string; hasAudio: boolean }>,
141
+ ): void {
142
+ const silent = clips.filter((c) => !c.hasAudio).map((c) => c.name);
143
+ if (silent.length > 0) {
144
+ throw new Error(
145
+ `no audio stream in: ${silent.join(", ")} — produce cuts by silence, so ` +
146
+ "every clip in a folder concat needs one (a silent b-roll clip can't be concatenated this way).",
147
+ );
148
+ }
149
+ }
150
+
151
+ const ConcatManifestSchema = z.object({
152
+ sort: z.enum(["name", "mtime"]),
153
+ entries: z.array(
154
+ z.object({
155
+ name: z.string(),
156
+ mtimeMs: z.number(),
157
+ size: z.number(),
158
+ durationSec: z.number(),
159
+ }),
160
+ ),
161
+ });
162
+ type ConcatManifest = z.infer<typeof ConcatManifestSchema>;
163
+
164
+ /**
165
+ * The cached build is reusable only if EVERY current file is present in the
166
+ * manifest with the same size and mtime (a changed byte count or timestamp
167
+ * means the clip could have been re-exported), the counts match (a clip
168
+ * removed leaves no trace otherwise), and the sort mode is the one the
169
+ * manifest was built under (a `--sort` change reorders the clips, so a
170
+ * same-files cache is still the WRONG concat).
171
+ *
172
+ * Belt-and-suspenders alongside `folderManifestKey`: the workdir is now
173
+ * content-addressed too, so in practice a stale manifest can only be reached
174
+ * by a hash collision or a folder mutated mid-run — this is what catches
175
+ * either without trusting the hash alone.
176
+ */
177
+ function manifestStillValid(manifest: ConcatManifest, sort: "name" | "mtime", current: readonly ConcatEntry[]): boolean {
178
+ if (manifest.sort !== sort) return false;
179
+ if (manifest.entries.length !== current.length) return false;
180
+ const byName = new Map(manifest.entries.map((e) => [e.name, e]));
181
+ return current.every((e) => {
182
+ const prev = byName.get(e.name);
183
+ return prev !== undefined && prev.mtimeMs === e.mtimeMs && prev.size === e.size;
184
+ });
185
+ }
186
+
187
+ export interface FolderListing {
188
+ entries: ConcatEntry[];
189
+ /** Files skipped for not matching a video extension (dotfiles excluded). */
190
+ nonVideoCount: number;
191
+ }
192
+
193
+ /**
194
+ * Enumerate the video files directly inside `folder` (no recursion — a
195
+ * subfolder is out of scope, not "ignored", so it is never counted).
196
+ *
197
+ * Symlinks to a regular file are followed (`stat`, which resolves the link)
198
+ * rather than dropped — a folder of symlinks into another drive is a normal
199
+ * way to stage takes, and silently enumerating zero clips from it would be a
200
+ * worse surprise than the extra `stat` call. A broken symlink or a symlink to
201
+ * a directory stats as "not a file" and is skipped without counting, the same
202
+ * as a real subfolder. Dotfiles (`.DS_Store` and friends) are skipped
203
+ * entirely and never counted — they are not a folder content decision the
204
+ * user made, so reporting them as "non-video files ignored" would be noise.
205
+ */
206
+ export async function listFolderVideos(folder: string): Promise<FolderListing> {
207
+ const dirents = await readdir(folder, { withFileTypes: true });
208
+ let nonVideoCount = 0;
209
+ const entries: ConcatEntry[] = [];
210
+ for (const d of dirents) {
211
+ if (d.name.startsWith(".")) continue;
212
+ let isFile = d.isFile();
213
+ if (!isFile && d.isSymbolicLink()) {
214
+ try {
215
+ isFile = (await stat(join(folder, d.name))).isFile();
216
+ } catch {
217
+ isFile = false; // broken symlink — treated like a subfolder: skipped, not counted
218
+ }
219
+ }
220
+ if (!isFile) continue; // a real subfolder, or a symlink to one
221
+ const dot = d.name.lastIndexOf(".");
222
+ const ext = dot >= 0 ? d.name.slice(dot + 1).toLowerCase() : "";
223
+ if (!VIDEO_EXTENSION_SET.has(ext)) {
224
+ nonVideoCount++;
225
+ continue;
226
+ }
227
+ const st = await stat(join(folder, d.name));
228
+ entries.push({ name: d.name, mtimeMs: st.mtimeMs, size: st.size });
229
+ }
230
+ if (entries.length === 0) throw noVideoFilesError(folder);
231
+ return { entries, nonVideoCount };
232
+ }
233
+
234
+ export interface FolderConcatResult {
235
+ /** The intermediate file — hand this to the rest of produce. */
236
+ path: string;
237
+ /** In final concat order, for the "one line per clip" console report. */
238
+ clips: Array<{ name: string; durationSec: number }>;
239
+ /** Files skipped for not matching a video extension. */
240
+ nonVideoCount: number;
241
+ /** True when the existing `source-concat.mp4` was reused, not rebuilt. */
242
+ cached: boolean;
243
+ /** The concat's own total duration (ffprobe'd from the output). */
244
+ durationSec: number;
245
+ }
246
+
247
+ /**
248
+ * Order and concat `listing`'s clips into `<workDir>/source-concat.mp4`,
249
+ * caching on a manifest of names+sizes+mtimes so an unchanged folder skips
250
+ * the re-encode. `listing` comes from `listFolderVideos` — the caller
251
+ * enumerates once, up front, because it ALSO needs the listing to derive the
252
+ * workdir's content-addressed hash (`folderManifestKey`) before this can
253
+ * even be called with a `workDir` to write into.
254
+ */
255
+ export async function concatFolder(
256
+ tools: IngestTools,
257
+ folder: string,
258
+ listing: FolderListing,
259
+ workDir: string,
260
+ sort: "name" | "mtime",
261
+ target: { w: number; h: number },
262
+ ): Promise<FolderConcatResult> {
263
+ const { entries: current, nonVideoCount } = listing;
264
+ if (current.length === 0) throw noVideoFilesError(folder);
265
+ const order = planFolderConcat(current, sort);
266
+
267
+ const outPath = join(workDir, "source-concat.mp4");
268
+ const manifestPath = join(workDir, "source-concat.json");
269
+ if (existsSync(outPath) && existsSync(manifestPath)) {
270
+ const parsed = ConcatManifestSchema.safeParse(JSON.parse(await readFile(manifestPath, "utf8")));
271
+ if (parsed.success && manifestStillValid(parsed.data, sort, current)) {
272
+ const byName = new Map(parsed.data.entries.map((e) => [e.name, e]));
273
+ const outProbe = await probe(tools, outPath);
274
+ return {
275
+ path: outPath,
276
+ clips: order.map((name) => ({ name, durationSec: byName.get(name)!.durationSec })),
277
+ nonVideoCount,
278
+ cached: true,
279
+ durationSec: outProbe.duration,
280
+ };
281
+ }
282
+ }
283
+
284
+ // A file with a video EXTENSION that fails to probe (no video stream) is an
285
+ // error naming the file, not a silent skip (folder-input-brief.md) — so
286
+ // `probe()`'s own "no video stream in <path>" is left to propagate rather
287
+ // than caught here.
288
+ const probes = await Promise.all(order.map((name) => probe(tools, join(folder, name))));
289
+ assertAllClipsHaveAudio(order.map((name, i) => ({ name, hasAudio: probes[i]!.hasAudio })));
290
+
291
+ const filter = buildConcatFilter(order.length, target);
292
+ const inputArgs = order.flatMap((name) => ["-i", join(folder, name)]);
293
+ // Encode to a sibling temp path, rename only on success — same reasoning as
294
+ // `bakeNormalizedSource` in normalize.ts (R27 §125): ffmpeg writes the
295
+ // container header as it goes, so a bake that dies mid-graph leaves a file
296
+ // with no `moov` atom, and a cache keyed on EXISTENCE would reuse that
297
+ // corpse forever. Rename is atomic on a POSIX filesystem.
298
+ const partial = `${outPath}.partial.mp4`;
299
+ try {
300
+ await run(tools.ffmpegPath, [
301
+ "-y",
302
+ ...inputArgs,
303
+ "-filter_complex", filter,
304
+ "-map", "[outv]",
305
+ "-map", "[outa]",
306
+ "-c:v", "libx264", "-preset", "medium", "-crf", "18",
307
+ "-c:a", "aac", "-b:a", "192k",
308
+ partial,
309
+ ]);
310
+ await rename(partial, outPath);
311
+ } catch (err) {
312
+ await rm(partial, { force: true });
313
+ throw err;
314
+ }
315
+
316
+ const manifest: ConcatManifest = {
317
+ sort,
318
+ entries: current.map((e) => ({
319
+ ...e,
320
+ durationSec: probes[order.indexOf(e.name)]!.duration,
321
+ })),
322
+ };
323
+ await writeFile(manifestPath, JSON.stringify(manifest, null, 2));
324
+
325
+ const outProbe = await probe(tools, outPath);
326
+ return {
327
+ path: outPath,
328
+ clips: order.map((name, i) => ({ name, durationSec: probes[i]!.duration })),
329
+ nonVideoCount,
330
+ cached: false,
331
+ durationSec: outProbe.duration,
332
+ };
333
+ }
package/src/cutlist.ts CHANGED
@@ -63,10 +63,19 @@ export interface BuildCutlistArgs {
63
63
  /**
64
64
  * Spans the speaker marked as bloopers out loud (R27 §122), from
65
65
  * `findBloopSpans`. Passed in rather than detected here so this stays a pure
66
- * function of its arguments — and so `--blooper-marker` is the only thing
67
- * that can put a `retake` cut in the timeline.
66
+ * function of its arguments.
68
67
  */
69
68
  bloops?: readonly { startWord: number; endWord: number; startSec: number; endSec: number }[];
69
+ /**
70
+ * Spans `findRetakeGroups` (R27 §128) elected to cut — the deterministic
71
+ * "keep only the last complete take" detector for the flub the speaker did
72
+ * NOT mark. Also a `reason: "retake"` cut, and also passed in rather than
73
+ * detected here, for the same purity reason as `bloops`: `buildCutlist`
74
+ * still has no judgement of its own about what a bad take looks like, it
75
+ * just folds whichever spans two independent detectors handed it into the
76
+ * one partition.
77
+ */
78
+ retakes?: readonly { startWord: number; endWord: number; startSec: number; endSec: number }[];
70
79
  }
71
80
 
72
81
  export function buildCutlist({
@@ -75,11 +84,13 @@ export function buildCutlist({
75
84
  duration,
76
85
  level,
77
86
  bloops,
87
+ retakes,
78
88
  }: BuildCutlistArgs): Segment[] {
79
89
  const keepAll: Segment[] = [{ srcIn: 0, srcOut: duration, kind: "keep" }];
80
90
  // `exact` means exact: it is the escape hatch for "touch nothing", and a
81
- // blooper cut is still a cut. --blooper-marker with --cleanup exact is a
82
- // contradiction, and the flag the user typed second does not get to win.
91
+ // blooper or retake cut is still a cut. --blooper-marker or
92
+ // --collapse-retakes with --cleanup exact is a contradiction, and the flag
93
+ // the user typed second does not get to win.
83
94
  if (level === "exact") return keepAll;
84
95
  const policy = POLICIES[level];
85
96
  const words = transcript.words;
@@ -105,6 +116,21 @@ export function buildCutlist({
105
116
  });
106
117
  }
107
118
 
119
+ // Same injection, same reason, lower confidence (R27 §128): a marker is the
120
+ // speaker asserting "this attempt is bad" — confidence 1. A retake group is
121
+ // this codebase inferring it from token similarity and the hallucination
122
+ // guard, so it earns 0.9, not 1 — the report and any future confidence-
123
+ // gated behavior can tell a supplied fact from an inferred one.
124
+ for (const r of retakes ?? []) {
125
+ removals.push({
126
+ start: r.startSec,
127
+ end: r.endSec,
128
+ reason: "retake",
129
+ confidence: 0.9,
130
+ source: "acoustic",
131
+ });
132
+ }
133
+
108
134
  for (const pause of analysis.cuttable) {
109
135
  // Lead and tail are decided by the SILENCE's position in the file, not by
110
136
  // comparing it to a word stamp (R27 §127). Whisper's `-ml 1` stamps stretch
@@ -178,7 +204,23 @@ export function buildCutlist({
178
204
  // keep. Acoustic boundaries are exempt: whisper's `-ml 1` stamps stretch a
179
205
  // word's end all the way to the next word's start, so a pause *always* looks
180
206
  // like it is "inside" a word — applying this rule to them cancels every cut.
181
- const protectedWords = words.filter((_, i) => !fillerIndices.has(i));
207
+ //
208
+ // Fillers are excluded here ONLY when `policy.removeFillers` says they're
209
+ // actually being removed (fix wave final review, findings §124's
210
+ // follow-up): this same `protectedWords` list also feeds
211
+ // `hasProtectedWordInside` below, which Task 6 widened to fold a wordless
212
+ // keep-gap up to `policy.pauseMin` (1.2s at light). Excluding fillers
213
+ // unconditionally meant a lone "um" sitting in a gap between two silence
214
+ // removals read as "wordless" even at `light`, where `removeFillers` is
215
+ // false and the filler was never scheduled for removal at all — so the
216
+ // fold silently ate it, cutting a word `light`'s own contract promises to
217
+ // keep. The transcript-boundary loop right below is unaffected: it only
218
+ // ever runs for `source === "transcript"` removals, which only exist when
219
+ // `policy.removeFillers` created them (the `if (policy.removeFillers)`
220
+ // block above) — so at `light`, that loop already sees zero such removals
221
+ // and this widened list changes nothing for it, verified by reading rather
222
+ // than assumed.
223
+ const protectedWords = words.filter((_, i) => !policy.removeFillers || !fillerIndices.has(i));
182
224
  for (const r of removals) {
183
225
  if (r.source !== "transcript") continue;
184
226
  for (const w of protectedWords) {
@@ -193,12 +235,40 @@ export function buildCutlist({
193
235
  return mid > start && mid < end;
194
236
  });
195
237
 
196
- // Merge removals that overlap, or whose in-between keep is a wordless sliver.
238
+ // Merge removals that overlap, or whose in-between keep is a wordless
239
+ // sliver — folded in regardless of length, not just when it's already
240
+ // under MIN_KEEP. A 0.37s wordless gap between two `silence` removals
241
+ // shipped in a real cleanup run because the old condition ANDed the
242
+ // wordless check to the length check, so `hasProtectedWordInside` was only
243
+ // ever asked once the gap was already short — a wordless gap that cleared
244
+ // MIN_KEEP was never asked at all (findings §124). MIN_KEEP's own comment
245
+ // already says wordless fragments fold; this makes the code do it.
246
+ //
247
+ // The fold is capped at `policy.pauseMin`, not left unbounded — folds any
248
+ // wordless gap UP TO pauseMin, refuses anything past it. The cap isn't
249
+ // about protecting short gaps; it's about what a gap LONGER than pauseMin
250
+ // sitting between two removals implies. The interior-pause branch above
251
+ // already generates its own removal for every genuinely silent stretch
252
+ // longer than pauseMin (`pauseDur <= policy.pauseMin` is the only case it
253
+ // skips) — so if a wordless-per-transcript gap that long survives here
254
+ // as bare space between two OTHER removals, the acoustic detector looked
255
+ // at it and did NOT call it silence. That's a live-audio signal the
256
+ // transcript can't see (a breath, laughter, room action, b-roll audio)
257
+ // being kept safe from a rule that only knows "the transcript found no
258
+ // words." A gap AT OR UNDER pauseMin, by contrast, is exactly the field
259
+ // bug's shape (0.37s, standard's 0.7s pauseMin): debris left over once
260
+ // both its neighbors are already cut, not a stretch the detector had any
261
+ // chance to flag on its own. `Math.max` with MIN_KEEP is defensive, not
262
+ // load-bearing: every current pauseMin already exceeds MIN_KEEP.
197
263
  const merged: Removal[] = [];
198
264
  for (const r of removals) {
199
265
  if (r.end - r.start < 0.05) continue;
200
266
  const prev = merged[merged.length - 1];
201
- if (prev && (r.start <= prev.end + 1e-6 || (r.start - prev.end < MIN_KEEP && !hasProtectedWordInside(prev.end, r.start)))) {
267
+ const gap = prev ? r.start - prev.end : Number.POSITIVE_INFINITY;
268
+ const overlapping = prev !== undefined && gap <= 1e-6;
269
+ const wordless = prev !== undefined && !hasProtectedWordInside(prev.end, r.start);
270
+ const foldableGap = wordless && gap <= Math.max(MIN_KEEP, policy.pauseMin);
271
+ if (prev && (overlapping || foldableGap)) {
202
272
  const prevDur = prev.end - prev.start;
203
273
  const curDur = r.end - r.start;
204
274
  prev.end = Math.max(prev.end, r.end);
package/src/index.ts CHANGED
@@ -6,12 +6,15 @@ export * from "./assemble";
6
6
  export * from "./fill";
7
7
  export * from "./producer/index";
8
8
  export * from "./timemap";
9
+ export * from "./recut";
9
10
  export * from "./ingest";
11
+ export * from "./concat";
10
12
  export * from "./transcribe";
11
13
  export * from "./analyze";
12
14
  export * from "./cutlist";
13
15
  export * from "./clip";
14
16
  export * from "./blooper";
17
+ export * from "./retake";
15
18
  export * from "./captions";
16
19
  export * from "./zoom";
17
20
  export * from "./grounding";