@ossclip/core 0.1.6 → 0.1.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/blooper.ts +82 -6
- package/src/concat.ts +333 -0
- package/src/cutlist.ts +77 -7
- package/src/index.ts +3 -0
- package/src/overrides.ts +157 -2
- package/src/phonetics.ts +6 -1
- package/src/recut.ts +335 -0
- package/src/retake.ts +607 -0
- package/src/scene-schema.ts +9 -1
- package/src/timemap.ts +37 -0
package/package.json
CHANGED
package/src/blooper.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { normalizeToken } from "./analyze";
|
|
2
2
|
import { isSentenceStart } from "./clip";
|
|
3
|
+
import { levenshtein } from "./phonetics";
|
|
3
4
|
import type { Transcript } from "./schema";
|
|
4
5
|
|
|
5
6
|
/**
|
|
@@ -14,8 +15,14 @@ import type { Transcript } from "./schema";
|
|
|
14
15
|
*
|
|
15
16
|
* A marker the speaker says OUT LOUD is the deterministic subset. It needs no
|
|
16
17
|
* judgement: the word is in the transcript or it is not. So this ships the
|
|
17
|
-
* useful half of the feature and leaves the guarantee intact
|
|
18
|
-
*
|
|
18
|
+
* useful half of the feature and leaves the guarantee intact.
|
|
19
|
+
*
|
|
20
|
+
* The OTHER half — the flub the speaker did NOT mark — turned out to have a
|
|
21
|
+
* deterministic formulation too: `retake.ts` (R27 §128) collapses consecutive
|
|
22
|
+
* near-identical sentences by token similarity, no LLM, same purity
|
|
23
|
+
* guarantee. What's left unbuilt is narrower than this comment used to claim:
|
|
24
|
+
* a genuinely REWORDED retake (different words, same idea) is still semantic,
|
|
25
|
+
* and stays out of scope on purpose (ROADMAP.md).
|
|
19
26
|
*
|
|
20
27
|
* The pattern, from the take that motivated it:
|
|
21
28
|
*
|
|
@@ -38,6 +45,57 @@ export interface BloopSpan {
|
|
|
38
45
|
endSec: number;
|
|
39
46
|
/** How many marker words this span swallowed — 2+ means repeated attempts. */
|
|
40
47
|
markers: number;
|
|
48
|
+
/** The marker text this span was searched for, normalized — for the report line. */
|
|
49
|
+
marker: string;
|
|
50
|
+
/**
|
|
51
|
+
* Surface forms in this span that matched by sound-alike or edit distance,
|
|
52
|
+
* not exact text — e.g. ASR wrote "looker" for a "blooper" marker. Every
|
|
53
|
+
* fuzzy hit must land here: it is what makes fuzzy matching safe to ship
|
|
54
|
+
* on by default, since a false positive shows up in report.txt instead of
|
|
55
|
+
* silently cutting a good take (Task 3, editor-dogfood-fixes plan).
|
|
56
|
+
*/
|
|
57
|
+
matched: string[];
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// Fuzzy matching only turns on once the marker is long enough that a false
|
|
61
|
+
// positive is unlikely — a short marker like "cut" sound-alikes ("cat") and
|
|
62
|
+
// sits within edit distance 2 of half the dictionary ("but", "gut", "cot"),
|
|
63
|
+
// so short markers stay exact-only (Task 3, editor-dogfood-fixes plan).
|
|
64
|
+
const FUZZY_MIN_MARKER_LEN = 6;
|
|
65
|
+
// "blooper" → "looker" is exactly this: 2 edits (drop the "b", substitute
|
|
66
|
+
// "p" for "k"). Found in the wild — see the guard test in blooper.test.ts.
|
|
67
|
+
const FUZZY_MAX_DISTANCE = 2;
|
|
68
|
+
|
|
69
|
+
interface MarkerMatch {
|
|
70
|
+
/** Normalized text of the transcript word that matched. */
|
|
71
|
+
surface: string;
|
|
72
|
+
/** False when this needed sound-alike/edit-distance rather than an exact hit. */
|
|
73
|
+
exact: boolean;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Whether a transcript word counts as the marker, and how.
|
|
78
|
+
*
|
|
79
|
+
* Exact match (today's rule) always wins first. Past that, the only fuzzy
|
|
80
|
+
* arm is a small edit distance — NOT `soundsSimilar` (§125,
|
|
81
|
+
* PHASE1-FINDINGS.md). The first field run of this feature paired the two
|
|
82
|
+
* arms as designed and got the worst of both: `soundsSimilar("builds",
|
|
83
|
+
* "blooper")` is true (shared "b" onset, score over its 0.34 floor) and cut
|
|
84
|
+
* 86.8% of a 125.9s video, while the pair `soundsSimilar` exists to catch —
|
|
85
|
+
* "looker" for "blooper" — is REJECTED by its own onset test (b/l differ)
|
|
86
|
+
* and only ever matched via Levenshtein anyway. Sound-alike was admitting
|
|
87
|
+
* garbage and catching nothing real, so it is gone; Levenshtein alone still
|
|
88
|
+
* catches "looker" (distance 2) and does not catch "builds" (distance 6).
|
|
89
|
+
*/
|
|
90
|
+
function matchMarker(wordText: string, want: string): MarkerMatch | null {
|
|
91
|
+
const norm = normalizeToken(wordText);
|
|
92
|
+
if (!norm) return null;
|
|
93
|
+
if (norm === want) return { surface: norm, exact: true };
|
|
94
|
+
if (want.length < FUZZY_MIN_MARKER_LEN) return null;
|
|
95
|
+
if (levenshtein(norm, want) <= FUZZY_MAX_DISTANCE) {
|
|
96
|
+
return { surface: norm, exact: false };
|
|
97
|
+
}
|
|
98
|
+
return null;
|
|
41
99
|
}
|
|
42
100
|
|
|
43
101
|
/**
|
|
@@ -45,7 +103,9 @@ export interface BloopSpan {
|
|
|
45
103
|
*
|
|
46
104
|
* `marker` is matched with `normalizeToken`, the same normalizer the filler
|
|
47
105
|
* detector uses, so case and trailing punctuation do not matter — ASR writes
|
|
48
|
-
* the word as "blooper." with the period riding on it.
|
|
106
|
+
* the word as "blooper." with the period riding on it. Beyond exact text, a
|
|
107
|
+
* marker of at least `FUZZY_MIN_MARKER_LEN` characters also matches an ASR
|
|
108
|
+
* mishearing — see `matchMarker`.
|
|
49
109
|
*
|
|
50
110
|
* Returns spans in transcript order, non-overlapping.
|
|
51
111
|
*/
|
|
@@ -53,11 +113,15 @@ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpa
|
|
|
53
113
|
const want = normalizeToken(marker);
|
|
54
114
|
if (!want) return [];
|
|
55
115
|
const words = transcript.words;
|
|
56
|
-
const
|
|
116
|
+
const matchAt = (i: number): MarkerMatch | null => {
|
|
117
|
+
const w = words[i];
|
|
118
|
+
return w ? matchMarker(w.text, want) : null;
|
|
119
|
+
};
|
|
57
120
|
|
|
58
121
|
const spans: BloopSpan[] = [];
|
|
59
122
|
for (let i = 0; i < words.length; i++) {
|
|
60
|
-
|
|
123
|
+
const match = matchAt(i);
|
|
124
|
+
if (!match) continue;
|
|
61
125
|
|
|
62
126
|
// Walk back over the attempt this marker spoiled, to the start of its
|
|
63
127
|
// sentence. The marker's own text usually ENDS a sentence ("blooper."), so
|
|
@@ -70,10 +134,12 @@ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpa
|
|
|
70
134
|
// one-word island of a sentence nobody finished.
|
|
71
135
|
const prev = spans[spans.length - 1];
|
|
72
136
|
let markers = 1;
|
|
137
|
+
let matched = match.exact ? [] : [match.surface];
|
|
73
138
|
if (prev && start <= prev.endWord + 1) {
|
|
74
139
|
spans.pop();
|
|
75
140
|
start = prev.startWord;
|
|
76
141
|
markers = prev.markers + 1;
|
|
142
|
+
matched = [...prev.matched, ...matched];
|
|
77
143
|
}
|
|
78
144
|
|
|
79
145
|
spans.push({
|
|
@@ -82,6 +148,8 @@ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpa
|
|
|
82
148
|
startSec: words[start]!.start,
|
|
83
149
|
endSec: words[i]!.end,
|
|
84
150
|
markers,
|
|
151
|
+
marker: want,
|
|
152
|
+
matched,
|
|
85
153
|
});
|
|
86
154
|
}
|
|
87
155
|
return spans;
|
|
@@ -98,5 +166,13 @@ export function formatBloopSpan(transcript: Transcript, span: BloopSpan): string
|
|
|
98
166
|
.map((w) => w.text)
|
|
99
167
|
.join(" ");
|
|
100
168
|
const attempts = span.markers > 1 ? ` (${span.markers} attempts)` : "";
|
|
101
|
-
|
|
169
|
+
// A fuzzy hit must never be silent — this line is the safety net that
|
|
170
|
+
// makes on-by-default fuzzy matching acceptable (Task 3, editor-dogfood-fixes
|
|
171
|
+
// plan): a false positive shows up here instead of quietly cutting a good
|
|
172
|
+
// take.
|
|
173
|
+
const fuzzy =
|
|
174
|
+
span.matched.length > 0
|
|
175
|
+
? " " + span.matched.map((m) => `matched "${m}" ~ "${span.marker}"`).join(", ")
|
|
176
|
+
: "";
|
|
177
|
+
return `"${said}"${attempts}${fuzzy}`;
|
|
102
178
|
}
|
package/src/concat.ts
ADDED
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
import { existsSync } from "node:fs";
|
|
2
|
+
import { readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
import { z } from "zod/v4";
|
|
5
|
+
import { probe, type IngestTools } from "./ingest";
|
|
6
|
+
import { run } from "./exec";
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* `ossclip produce <folder>` (2026-08-05 field request, verbatim intent in
|
|
10
|
+
* .superpowers/sdd/folder-input-brief.md): a folder of camera-clip takes gets
|
|
11
|
+
* concatenated into ONE source before the normal produce pipeline ever sees
|
|
12
|
+
* it, instead of the user hand-concatenating with an agent first.
|
|
13
|
+
*
|
|
14
|
+
* Split per CLAUDE.md's pure/IO mandate — the bug that motivates the split is
|
|
15
|
+
* in the brief: a hand-built ffmpeg filtergraph corrupted by shell expansion
|
|
16
|
+
* (zsh's `:a` history modifier) silently played clip 1's audio in every
|
|
17
|
+
* concat slot, and nothing validated the filter STRING before it reached
|
|
18
|
+
* ffmpeg. `planFolderConcat`, `buildConcatFilter`, `folderManifestKey` and
|
|
19
|
+
* `assertAllClipsHaveAudio` are pure so each can be asserted on directly;
|
|
20
|
+
* `listFolderVideos` and `concatFolder` are the only I/O.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
const VIDEO_EXTENSIONS = ["mov", "mp4", "m4v", "mkv", "webm", "avi"] as const;
|
|
24
|
+
const VIDEO_EXTENSION_SET = new Set<string>(VIDEO_EXTENSIONS);
|
|
25
|
+
|
|
26
|
+
function noVideoFilesError(folder: string): Error {
|
|
27
|
+
return new Error(
|
|
28
|
+
`no video files found directly inside ${folder} ` +
|
|
29
|
+
`(looked for: ${VIDEO_EXTENSIONS.map((e) => `.${e}`).join(", ")})`,
|
|
30
|
+
);
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export interface ConcatEntry {
|
|
34
|
+
name: string;
|
|
35
|
+
mtimeMs: number;
|
|
36
|
+
size: number;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Order clips for concatenation. `name` (default, per the field request "sort
|
|
41
|
+
* them by name or date modified, name being default") is a PLAIN codepoint
|
|
42
|
+
* sort — comparing strings with `<`/`>` rather than `localeCompare`, which is
|
|
43
|
+
* what `ls` gives on a case-sensitive filesystem and a locale-aware sort would
|
|
44
|
+
* NOT: it reorders case and punctuation differently per machine locale, which
|
|
45
|
+
* would make the same folder concat in a different order on a different
|
|
46
|
+
* machine. `mtime` ties (a batch copy that preserved one timestamp across
|
|
47
|
+
* several files) fall back to name so the order stays reproducible either way.
|
|
48
|
+
*/
|
|
49
|
+
export function planFolderConcat(
|
|
50
|
+
entries: readonly ConcatEntry[],
|
|
51
|
+
sort: "name" | "mtime",
|
|
52
|
+
): string[] {
|
|
53
|
+
const byName = (a: ConcatEntry, b: ConcatEntry): number =>
|
|
54
|
+
a.name < b.name ? -1 : a.name > b.name ? 1 : 0;
|
|
55
|
+
const cmp = sort === "name" ? byName : (a: ConcatEntry, b: ConcatEntry) => a.mtimeMs - b.mtimeMs || byName(a, b);
|
|
56
|
+
return [...entries].sort(cmp).map((e) => e.name);
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* A deterministic, order-independent identity for a folder's clip set —
|
|
61
|
+
* sorted by name (a canonical order regardless of `readdir`'s OS-dependent
|
|
62
|
+
* enumeration order) before joining, so the same files always hash the same
|
|
63
|
+
* way. `sort` is folded in because a `--sort` flip changes the concat's
|
|
64
|
+
* actual bytes (different clip order), not just how it was chosen.
|
|
65
|
+
*
|
|
66
|
+
* Fix for a review finding on the first cut of this feature: the workdir
|
|
67
|
+
* hash used to be derived from the FOLDER PATH alone, which is stable across
|
|
68
|
+
* content changes — but `audio.wav`, `transcript.json`, the content-rect
|
|
69
|
+
* cache and the mezzanine are all existence-keyed inside that same workdir.
|
|
70
|
+
* Adding a take (or flipping --sort) rebuilt `source-concat.mp4` correctly
|
|
71
|
+
* but silently reused every one of those, producing a video with captions
|
|
72
|
+
* transcribed against the PREVIOUS concat. Hashing the manifest content here
|
|
73
|
+
* — the same invariant a file input already has via `sha1File` — means a
|
|
74
|
+
* changed folder gets a fresh workdir, and every derived cache is fresh too.
|
|
75
|
+
*
|
|
76
|
+
* Serialized with JSON.stringify, not a `:`/`|` delimiter join (audit fix):
|
|
77
|
+
* a filename is user-controlled free text that can itself contain the
|
|
78
|
+
* delimiters, letting two DIFFERENT entry sets serialize to one identical
|
|
79
|
+
* key — `a:1` sized 2 and `a` sized `1:2` collide under a `:` join, and a
|
|
80
|
+
* collision here means one folder silently reuses another's transcript and
|
|
81
|
+
* mezzanine. JSON escapes the filename instead of trusting it. This changed
|
|
82
|
+
* every existing folder workdir hash once — a one-time cache invalidation
|
|
83
|
+
* (fresh workdir, full re-concat/re-transcribe on the next run), accepted as
|
|
84
|
+
* the cost of an injection-proof key.
|
|
85
|
+
*/
|
|
86
|
+
export function folderManifestKey(entries: readonly ConcatEntry[], sort: "name" | "mtime"): string {
|
|
87
|
+
const canonical = [...entries]
|
|
88
|
+
.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0))
|
|
89
|
+
.map((e) => ({ name: e.name, size: e.size, mtimeMs: e.mtimeMs }));
|
|
90
|
+
return JSON.stringify({ sort, entries: canonical });
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* The `-filter_complex` string concatenating `n` inputs into one output.
|
|
95
|
+
* Each input gets its OWN scale+pad+fps+setsar+format chain — letterboxed to
|
|
96
|
+
* `target`, never cropped, since produce's own framing decides crops later —
|
|
97
|
+
* and its own audio resample, before the `concat` filter joins them.
|
|
98
|
+
*
|
|
99
|
+
* Rotation: ffmpeg auto-rotates on decode by default (R27 §119, `probe()` in
|
|
100
|
+
* ingest.ts relies on the same fact), so every `[i:v]` input here is ASSUMED
|
|
101
|
+
* to already arrive in DISPLAYED orientation — the scale/pad math needs no
|
|
102
|
+
* separate rotation step. This is an assumption carried over from R27 §119,
|
|
103
|
+
* not something the folder-input verification run could independently
|
|
104
|
+
* confirm: a letterboxed, correctly-proportioned 1080x1920 output is also
|
|
105
|
+
* what a WRONG rotation assumption would produce once padded to a portrait
|
|
106
|
+
* canvas, so that run couldn't distinguish "handled correctly" from
|
|
107
|
+
* "accidentally looks fine." Flagged rather than overclaimed per CLAUDE.md.
|
|
108
|
+
*
|
|
109
|
+
* `n` labels of each kind, never more or fewer, and the tail's `[vI][aI]`
|
|
110
|
+
* pairs are built in the SAME loop that emits them — this is the direct
|
|
111
|
+
* regression test target for the field bug (see module comment): a
|
|
112
|
+
* hand-built graph had a slot silently reference input 0's audio a second
|
|
113
|
+
* time instead of its own index, and nothing caught the STRING being wrong.
|
|
114
|
+
*/
|
|
115
|
+
export function buildConcatFilter(n: number, target: { w: number; h: number }): string {
|
|
116
|
+
const chains: string[] = [];
|
|
117
|
+
const tail: string[] = [];
|
|
118
|
+
for (let i = 0; i < n; i++) {
|
|
119
|
+
chains.push(
|
|
120
|
+
`[${i}:v]scale=${target.w}:${target.h}:force_original_aspect_ratio=decrease,` +
|
|
121
|
+
`pad=${target.w}:${target.h}:(ow-iw)/2:(oh-ih)/2,fps=30,setsar=1,format=yuv420p[v${i}]`,
|
|
122
|
+
);
|
|
123
|
+
chains.push(`[${i}:a]aresample=48000,aformat=channel_layouts=stereo[a${i}]`);
|
|
124
|
+
tail.push(`[v${i}][a${i}]`);
|
|
125
|
+
}
|
|
126
|
+
return `${chains.join(";")};${tail.join("")}concat=n=${n}:v=1:a=1[outv][outa]`;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* `buildConcatFilter` emits `[i:a]` unconditionally for every input — it has
|
|
131
|
+
* no way to know a clip is silent. Handed a video-only clip (b-roll with no
|
|
132
|
+
* audio stream), ffmpeg dies deep inside the filtergraph with a bare stream-
|
|
133
|
+
* specifier error that names neither the clip nor the reason. `probe()`
|
|
134
|
+
* already reports `hasAudio`; failing HERE, before ffmpeg ever runs, keeps
|
|
135
|
+
* faith with the brief's "a file that probe() rejects is an error naming the
|
|
136
|
+
* file, not a silent skip" — a clip with no audio is the same class of
|
|
137
|
+
* problem, just discovered an instant later than "no video stream at all".
|
|
138
|
+
*/
|
|
139
|
+
export function assertAllClipsHaveAudio(
|
|
140
|
+
clips: ReadonlyArray<{ name: string; hasAudio: boolean }>,
|
|
141
|
+
): void {
|
|
142
|
+
const silent = clips.filter((c) => !c.hasAudio).map((c) => c.name);
|
|
143
|
+
if (silent.length > 0) {
|
|
144
|
+
throw new Error(
|
|
145
|
+
`no audio stream in: ${silent.join(", ")} — produce cuts by silence, so ` +
|
|
146
|
+
"every clip in a folder concat needs one (a silent b-roll clip can't be concatenated this way).",
|
|
147
|
+
);
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
const ConcatManifestSchema = z.object({
|
|
152
|
+
sort: z.enum(["name", "mtime"]),
|
|
153
|
+
entries: z.array(
|
|
154
|
+
z.object({
|
|
155
|
+
name: z.string(),
|
|
156
|
+
mtimeMs: z.number(),
|
|
157
|
+
size: z.number(),
|
|
158
|
+
durationSec: z.number(),
|
|
159
|
+
}),
|
|
160
|
+
),
|
|
161
|
+
});
|
|
162
|
+
type ConcatManifest = z.infer<typeof ConcatManifestSchema>;
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* The cached build is reusable only if EVERY current file is present in the
|
|
166
|
+
* manifest with the same size and mtime (a changed byte count or timestamp
|
|
167
|
+
* means the clip could have been re-exported), the counts match (a clip
|
|
168
|
+
* removed leaves no trace otherwise), and the sort mode is the one the
|
|
169
|
+
* manifest was built under (a `--sort` change reorders the clips, so a
|
|
170
|
+
* same-files cache is still the WRONG concat).
|
|
171
|
+
*
|
|
172
|
+
* Belt-and-suspenders alongside `folderManifestKey`: the workdir is now
|
|
173
|
+
* content-addressed too, so in practice a stale manifest can only be reached
|
|
174
|
+
* by a hash collision or a folder mutated mid-run — this is what catches
|
|
175
|
+
* either without trusting the hash alone.
|
|
176
|
+
*/
|
|
177
|
+
function manifestStillValid(manifest: ConcatManifest, sort: "name" | "mtime", current: readonly ConcatEntry[]): boolean {
|
|
178
|
+
if (manifest.sort !== sort) return false;
|
|
179
|
+
if (manifest.entries.length !== current.length) return false;
|
|
180
|
+
const byName = new Map(manifest.entries.map((e) => [e.name, e]));
|
|
181
|
+
return current.every((e) => {
|
|
182
|
+
const prev = byName.get(e.name);
|
|
183
|
+
return prev !== undefined && prev.mtimeMs === e.mtimeMs && prev.size === e.size;
|
|
184
|
+
});
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
export interface FolderListing {
|
|
188
|
+
entries: ConcatEntry[];
|
|
189
|
+
/** Files skipped for not matching a video extension (dotfiles excluded). */
|
|
190
|
+
nonVideoCount: number;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* Enumerate the video files directly inside `folder` (no recursion — a
|
|
195
|
+
* subfolder is out of scope, not "ignored", so it is never counted).
|
|
196
|
+
*
|
|
197
|
+
* Symlinks to a regular file are followed (`stat`, which resolves the link)
|
|
198
|
+
* rather than dropped — a folder of symlinks into another drive is a normal
|
|
199
|
+
* way to stage takes, and silently enumerating zero clips from it would be a
|
|
200
|
+
* worse surprise than the extra `stat` call. A broken symlink or a symlink to
|
|
201
|
+
* a directory stats as "not a file" and is skipped without counting, the same
|
|
202
|
+
* as a real subfolder. Dotfiles (`.DS_Store` and friends) are skipped
|
|
203
|
+
* entirely and never counted — they are not a folder content decision the
|
|
204
|
+
* user made, so reporting them as "non-video files ignored" would be noise.
|
|
205
|
+
*/
|
|
206
|
+
export async function listFolderVideos(folder: string): Promise<FolderListing> {
|
|
207
|
+
const dirents = await readdir(folder, { withFileTypes: true });
|
|
208
|
+
let nonVideoCount = 0;
|
|
209
|
+
const entries: ConcatEntry[] = [];
|
|
210
|
+
for (const d of dirents) {
|
|
211
|
+
if (d.name.startsWith(".")) continue;
|
|
212
|
+
let isFile = d.isFile();
|
|
213
|
+
if (!isFile && d.isSymbolicLink()) {
|
|
214
|
+
try {
|
|
215
|
+
isFile = (await stat(join(folder, d.name))).isFile();
|
|
216
|
+
} catch {
|
|
217
|
+
isFile = false; // broken symlink — treated like a subfolder: skipped, not counted
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
if (!isFile) continue; // a real subfolder, or a symlink to one
|
|
221
|
+
const dot = d.name.lastIndexOf(".");
|
|
222
|
+
const ext = dot >= 0 ? d.name.slice(dot + 1).toLowerCase() : "";
|
|
223
|
+
if (!VIDEO_EXTENSION_SET.has(ext)) {
|
|
224
|
+
nonVideoCount++;
|
|
225
|
+
continue;
|
|
226
|
+
}
|
|
227
|
+
const st = await stat(join(folder, d.name));
|
|
228
|
+
entries.push({ name: d.name, mtimeMs: st.mtimeMs, size: st.size });
|
|
229
|
+
}
|
|
230
|
+
if (entries.length === 0) throw noVideoFilesError(folder);
|
|
231
|
+
return { entries, nonVideoCount };
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
export interface FolderConcatResult {
|
|
235
|
+
/** The intermediate file — hand this to the rest of produce. */
|
|
236
|
+
path: string;
|
|
237
|
+
/** In final concat order, for the "one line per clip" console report. */
|
|
238
|
+
clips: Array<{ name: string; durationSec: number }>;
|
|
239
|
+
/** Files skipped for not matching a video extension. */
|
|
240
|
+
nonVideoCount: number;
|
|
241
|
+
/** True when the existing `source-concat.mp4` was reused, not rebuilt. */
|
|
242
|
+
cached: boolean;
|
|
243
|
+
/** The concat's own total duration (ffprobe'd from the output). */
|
|
244
|
+
durationSec: number;
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
/**
|
|
248
|
+
* Order and concat `listing`'s clips into `<workDir>/source-concat.mp4`,
|
|
249
|
+
* caching on a manifest of names+sizes+mtimes so an unchanged folder skips
|
|
250
|
+
* the re-encode. `listing` comes from `listFolderVideos` — the caller
|
|
251
|
+
* enumerates once, up front, because it ALSO needs the listing to derive the
|
|
252
|
+
* workdir's content-addressed hash (`folderManifestKey`) before this can
|
|
253
|
+
* even be called with a `workDir` to write into.
|
|
254
|
+
*/
|
|
255
|
+
export async function concatFolder(
|
|
256
|
+
tools: IngestTools,
|
|
257
|
+
folder: string,
|
|
258
|
+
listing: FolderListing,
|
|
259
|
+
workDir: string,
|
|
260
|
+
sort: "name" | "mtime",
|
|
261
|
+
target: { w: number; h: number },
|
|
262
|
+
): Promise<FolderConcatResult> {
|
|
263
|
+
const { entries: current, nonVideoCount } = listing;
|
|
264
|
+
if (current.length === 0) throw noVideoFilesError(folder);
|
|
265
|
+
const order = planFolderConcat(current, sort);
|
|
266
|
+
|
|
267
|
+
const outPath = join(workDir, "source-concat.mp4");
|
|
268
|
+
const manifestPath = join(workDir, "source-concat.json");
|
|
269
|
+
if (existsSync(outPath) && existsSync(manifestPath)) {
|
|
270
|
+
const parsed = ConcatManifestSchema.safeParse(JSON.parse(await readFile(manifestPath, "utf8")));
|
|
271
|
+
if (parsed.success && manifestStillValid(parsed.data, sort, current)) {
|
|
272
|
+
const byName = new Map(parsed.data.entries.map((e) => [e.name, e]));
|
|
273
|
+
const outProbe = await probe(tools, outPath);
|
|
274
|
+
return {
|
|
275
|
+
path: outPath,
|
|
276
|
+
clips: order.map((name) => ({ name, durationSec: byName.get(name)!.durationSec })),
|
|
277
|
+
nonVideoCount,
|
|
278
|
+
cached: true,
|
|
279
|
+
durationSec: outProbe.duration,
|
|
280
|
+
};
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
// A file with a video EXTENSION that fails to probe (no video stream) is an
|
|
285
|
+
// error naming the file, not a silent skip (folder-input-brief.md) — so
|
|
286
|
+
// `probe()`'s own "no video stream in <path>" is left to propagate rather
|
|
287
|
+
// than caught here.
|
|
288
|
+
const probes = await Promise.all(order.map((name) => probe(tools, join(folder, name))));
|
|
289
|
+
assertAllClipsHaveAudio(order.map((name, i) => ({ name, hasAudio: probes[i]!.hasAudio })));
|
|
290
|
+
|
|
291
|
+
const filter = buildConcatFilter(order.length, target);
|
|
292
|
+
const inputArgs = order.flatMap((name) => ["-i", join(folder, name)]);
|
|
293
|
+
// Encode to a sibling temp path, rename only on success — same reasoning as
|
|
294
|
+
// `bakeNormalizedSource` in normalize.ts (R27 §125): ffmpeg writes the
|
|
295
|
+
// container header as it goes, so a bake that dies mid-graph leaves a file
|
|
296
|
+
// with no `moov` atom, and a cache keyed on EXISTENCE would reuse that
|
|
297
|
+
// corpse forever. Rename is atomic on a POSIX filesystem.
|
|
298
|
+
const partial = `${outPath}.partial.mp4`;
|
|
299
|
+
try {
|
|
300
|
+
await run(tools.ffmpegPath, [
|
|
301
|
+
"-y",
|
|
302
|
+
...inputArgs,
|
|
303
|
+
"-filter_complex", filter,
|
|
304
|
+
"-map", "[outv]",
|
|
305
|
+
"-map", "[outa]",
|
|
306
|
+
"-c:v", "libx264", "-preset", "medium", "-crf", "18",
|
|
307
|
+
"-c:a", "aac", "-b:a", "192k",
|
|
308
|
+
partial,
|
|
309
|
+
]);
|
|
310
|
+
await rename(partial, outPath);
|
|
311
|
+
} catch (err) {
|
|
312
|
+
await rm(partial, { force: true });
|
|
313
|
+
throw err;
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
const manifest: ConcatManifest = {
|
|
317
|
+
sort,
|
|
318
|
+
entries: current.map((e) => ({
|
|
319
|
+
...e,
|
|
320
|
+
durationSec: probes[order.indexOf(e.name)]!.duration,
|
|
321
|
+
})),
|
|
322
|
+
};
|
|
323
|
+
await writeFile(manifestPath, JSON.stringify(manifest, null, 2));
|
|
324
|
+
|
|
325
|
+
const outProbe = await probe(tools, outPath);
|
|
326
|
+
return {
|
|
327
|
+
path: outPath,
|
|
328
|
+
clips: order.map((name, i) => ({ name, durationSec: probes[i]!.duration })),
|
|
329
|
+
nonVideoCount,
|
|
330
|
+
cached: false,
|
|
331
|
+
durationSec: outProbe.duration,
|
|
332
|
+
};
|
|
333
|
+
}
|
package/src/cutlist.ts
CHANGED
|
@@ -63,10 +63,19 @@ export interface BuildCutlistArgs {
|
|
|
63
63
|
/**
|
|
64
64
|
* Spans the speaker marked as bloopers out loud (R27 §122), from
|
|
65
65
|
* `findBloopSpans`. Passed in rather than detected here so this stays a pure
|
|
66
|
-
* function of its arguments
|
|
67
|
-
* that can put a `retake` cut in the timeline.
|
|
66
|
+
* function of its arguments.
|
|
68
67
|
*/
|
|
69
68
|
bloops?: readonly { startWord: number; endWord: number; startSec: number; endSec: number }[];
|
|
69
|
+
/**
|
|
70
|
+
* Spans `findRetakeGroups` (R27 §128) elected to cut — the deterministic
|
|
71
|
+
* "keep only the last complete take" detector for the flub the speaker did
|
|
72
|
+
* NOT mark. Also a `reason: "retake"` cut, and also passed in rather than
|
|
73
|
+
* detected here, for the same purity reason as `bloops`: `buildCutlist`
|
|
74
|
+
* still has no judgement of its own about what a bad take looks like, it
|
|
75
|
+
* just folds whichever spans two independent detectors handed it into the
|
|
76
|
+
* one partition.
|
|
77
|
+
*/
|
|
78
|
+
retakes?: readonly { startWord: number; endWord: number; startSec: number; endSec: number }[];
|
|
70
79
|
}
|
|
71
80
|
|
|
72
81
|
export function buildCutlist({
|
|
@@ -75,11 +84,13 @@ export function buildCutlist({
|
|
|
75
84
|
duration,
|
|
76
85
|
level,
|
|
77
86
|
bloops,
|
|
87
|
+
retakes,
|
|
78
88
|
}: BuildCutlistArgs): Segment[] {
|
|
79
89
|
const keepAll: Segment[] = [{ srcIn: 0, srcOut: duration, kind: "keep" }];
|
|
80
90
|
// `exact` means exact: it is the escape hatch for "touch nothing", and a
|
|
81
|
-
// blooper cut is still a cut. --blooper-marker
|
|
82
|
-
//
|
|
91
|
+
// blooper or retake cut is still a cut. --blooper-marker or
|
|
92
|
+
// --collapse-retakes with --cleanup exact is a contradiction, and the flag
|
|
93
|
+
// the user typed second does not get to win.
|
|
83
94
|
if (level === "exact") return keepAll;
|
|
84
95
|
const policy = POLICIES[level];
|
|
85
96
|
const words = transcript.words;
|
|
@@ -105,6 +116,21 @@ export function buildCutlist({
|
|
|
105
116
|
});
|
|
106
117
|
}
|
|
107
118
|
|
|
119
|
+
// Same injection, same reason, lower confidence (R27 §128): a marker is the
|
|
120
|
+
// speaker asserting "this attempt is bad" — confidence 1. A retake group is
|
|
121
|
+
// this codebase inferring it from token similarity and the hallucination
|
|
122
|
+
// guard, so it earns 0.9, not 1 — the report and any future confidence-
|
|
123
|
+
// gated behavior can tell a supplied fact from an inferred one.
|
|
124
|
+
for (const r of retakes ?? []) {
|
|
125
|
+
removals.push({
|
|
126
|
+
start: r.startSec,
|
|
127
|
+
end: r.endSec,
|
|
128
|
+
reason: "retake",
|
|
129
|
+
confidence: 0.9,
|
|
130
|
+
source: "acoustic",
|
|
131
|
+
});
|
|
132
|
+
}
|
|
133
|
+
|
|
108
134
|
for (const pause of analysis.cuttable) {
|
|
109
135
|
// Lead and tail are decided by the SILENCE's position in the file, not by
|
|
110
136
|
// comparing it to a word stamp (R27 §127). Whisper's `-ml 1` stamps stretch
|
|
@@ -178,7 +204,23 @@ export function buildCutlist({
|
|
|
178
204
|
// keep. Acoustic boundaries are exempt: whisper's `-ml 1` stamps stretch a
|
|
179
205
|
// word's end all the way to the next word's start, so a pause *always* looks
|
|
180
206
|
// like it is "inside" a word — applying this rule to them cancels every cut.
|
|
181
|
-
|
|
207
|
+
//
|
|
208
|
+
// Fillers are excluded here ONLY when `policy.removeFillers` says they're
|
|
209
|
+
// actually being removed (fix wave final review, findings §124's
|
|
210
|
+
// follow-up): this same `protectedWords` list also feeds
|
|
211
|
+
// `hasProtectedWordInside` below, which Task 6 widened to fold a wordless
|
|
212
|
+
// keep-gap up to `policy.pauseMin` (1.2s at light). Excluding fillers
|
|
213
|
+
// unconditionally meant a lone "um" sitting in a gap between two silence
|
|
214
|
+
// removals read as "wordless" even at `light`, where `removeFillers` is
|
|
215
|
+
// false and the filler was never scheduled for removal at all — so the
|
|
216
|
+
// fold silently ate it, cutting a word `light`'s own contract promises to
|
|
217
|
+
// keep. The transcript-boundary loop right below is unaffected: it only
|
|
218
|
+
// ever runs for `source === "transcript"` removals, which only exist when
|
|
219
|
+
// `policy.removeFillers` created them (the `if (policy.removeFillers)`
|
|
220
|
+
// block above) — so at `light`, that loop already sees zero such removals
|
|
221
|
+
// and this widened list changes nothing for it, verified by reading rather
|
|
222
|
+
// than assumed.
|
|
223
|
+
const protectedWords = words.filter((_, i) => !policy.removeFillers || !fillerIndices.has(i));
|
|
182
224
|
for (const r of removals) {
|
|
183
225
|
if (r.source !== "transcript") continue;
|
|
184
226
|
for (const w of protectedWords) {
|
|
@@ -193,12 +235,40 @@ export function buildCutlist({
|
|
|
193
235
|
return mid > start && mid < end;
|
|
194
236
|
});
|
|
195
237
|
|
|
196
|
-
// Merge removals that overlap, or whose in-between keep is a wordless
|
|
238
|
+
// Merge removals that overlap, or whose in-between keep is a wordless
|
|
239
|
+
// sliver — folded in regardless of length, not just when it's already
|
|
240
|
+
// under MIN_KEEP. A 0.37s wordless gap between two `silence` removals
|
|
241
|
+
// shipped in a real cleanup run because the old condition ANDed the
|
|
242
|
+
// wordless check to the length check, so `hasProtectedWordInside` was only
|
|
243
|
+
// ever asked once the gap was already short — a wordless gap that cleared
|
|
244
|
+
// MIN_KEEP was never asked at all (findings §124). MIN_KEEP's own comment
|
|
245
|
+
// already says wordless fragments fold; this makes the code do it.
|
|
246
|
+
//
|
|
247
|
+
// The fold is capped at `policy.pauseMin`, not left unbounded — folds any
|
|
248
|
+
// wordless gap UP TO pauseMin, refuses anything past it. The cap isn't
|
|
249
|
+
// about protecting short gaps; it's about what a gap LONGER than pauseMin
|
|
250
|
+
// sitting between two removals implies. The interior-pause branch above
|
|
251
|
+
// already generates its own removal for every genuinely silent stretch
|
|
252
|
+
// longer than pauseMin (`pauseDur <= policy.pauseMin` is the only case it
|
|
253
|
+
// skips) — so if a wordless-per-transcript gap that long survives here
|
|
254
|
+
// as bare space between two OTHER removals, the acoustic detector looked
|
|
255
|
+
// at it and did NOT call it silence. That's a live-audio signal the
|
|
256
|
+
// transcript can't see (a breath, laughter, room action, b-roll audio)
|
|
257
|
+
// being kept safe from a rule that only knows "the transcript found no
|
|
258
|
+
// words." A gap AT OR UNDER pauseMin, by contrast, is exactly the field
|
|
259
|
+
// bug's shape (0.37s, standard's 0.7s pauseMin): debris left over once
|
|
260
|
+
// both its neighbors are already cut, not a stretch the detector had any
|
|
261
|
+
// chance to flag on its own. `Math.max` with MIN_KEEP is defensive, not
|
|
262
|
+
// load-bearing: every current pauseMin already exceeds MIN_KEEP.
|
|
197
263
|
const merged: Removal[] = [];
|
|
198
264
|
for (const r of removals) {
|
|
199
265
|
if (r.end - r.start < 0.05) continue;
|
|
200
266
|
const prev = merged[merged.length - 1];
|
|
201
|
-
|
|
267
|
+
const gap = prev ? r.start - prev.end : Number.POSITIVE_INFINITY;
|
|
268
|
+
const overlapping = prev !== undefined && gap <= 1e-6;
|
|
269
|
+
const wordless = prev !== undefined && !hasProtectedWordInside(prev.end, r.start);
|
|
270
|
+
const foldableGap = wordless && gap <= Math.max(MIN_KEEP, policy.pauseMin);
|
|
271
|
+
if (prev && (overlapping || foldableGap)) {
|
|
202
272
|
const prevDur = prev.end - prev.start;
|
|
203
273
|
const curDur = r.end - r.start;
|
|
204
274
|
prev.end = Math.max(prev.end, r.end);
|
package/src/index.ts
CHANGED
|
@@ -6,12 +6,15 @@ export * from "./assemble";
|
|
|
6
6
|
export * from "./fill";
|
|
7
7
|
export * from "./producer/index";
|
|
8
8
|
export * from "./timemap";
|
|
9
|
+
export * from "./recut";
|
|
9
10
|
export * from "./ingest";
|
|
11
|
+
export * from "./concat";
|
|
10
12
|
export * from "./transcribe";
|
|
11
13
|
export * from "./analyze";
|
|
12
14
|
export * from "./cutlist";
|
|
13
15
|
export * from "./clip";
|
|
14
16
|
export * from "./blooper";
|
|
17
|
+
export * from "./retake";
|
|
15
18
|
export * from "./captions";
|
|
16
19
|
export * from "./zoom";
|
|
17
20
|
export * from "./grounding";
|