@ossclip/core 0.1.7 → 0.1.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ossclip/core",
3
- "version": "0.1.7",
3
+ "version": "0.1.10",
4
4
  "description": "ossclip's framework-free pipeline: schema, transcription, analysis, cutlist, captions, framing, and the LLM producer",
5
5
  "type": "module",
6
6
  "license": "MIT",
package/src/blooper.ts CHANGED
@@ -1,5 +1,6 @@
1
1
  import { normalizeToken } from "./analyze";
2
2
  import { isSentenceStart } from "./clip";
3
+ import { levenshtein } from "./phonetics";
3
4
  import type { Transcript } from "./schema";
4
5
 
5
6
  /**
@@ -14,8 +15,14 @@ import type { Transcript } from "./schema";
14
15
  *
15
16
  * A marker the speaker says OUT LOUD is the deterministic subset. It needs no
16
17
  * judgement: the word is in the transcript or it is not. So this ships the
17
- * useful half of the feature and leaves the guarantee intact — the semantic
18
- * detector remains unbuilt, deliberately.
18
+ * useful half of the feature and leaves the guarantee intact.
19
+ *
20
+ * The OTHER half — the flub the speaker did NOT mark — turned out to have a
21
+ * deterministic formulation too: `retake.ts` (R27 §128) collapses consecutive
22
+ * near-identical sentences by token similarity, no LLM, same purity
23
+ * guarantee. What's left unbuilt is narrower than this comment used to claim:
24
+ * a genuinely REWORDED retake (different words, same idea) is still semantic,
25
+ * and stays out of scope on purpose (ROADMAP.md).
19
26
  *
20
27
  * The pattern, from the take that motivated it:
21
28
  *
@@ -38,6 +45,57 @@ export interface BloopSpan {
38
45
  endSec: number;
39
46
  /** How many marker words this span swallowed — 2+ means repeated attempts. */
40
47
  markers: number;
48
+ /** The marker text this span was searched for, normalized — for the report line. */
49
+ marker: string;
50
+ /**
51
+ * Surface forms in this span that matched by sound-alike or edit distance,
52
+ * not exact text — e.g. ASR wrote "looker" for a "blooper" marker. Every
53
+ * fuzzy hit must land here: it is what makes fuzzy matching safe to ship
54
+ * on by default, since a false positive shows up in report.txt instead of
55
+ * silently cutting a good take (Task 3, editor-dogfood-fixes plan).
56
+ */
57
+ matched: string[];
58
+ }
59
+
60
+ // Fuzzy matching only turns on once the marker is long enough that a false
61
+ // positive is unlikely — a short marker like "cut" sound-alikes ("cat") and
62
+ // sits within edit distance 2 of half the dictionary ("but", "gut", "cot"),
63
+ // so short markers stay exact-only (Task 3, editor-dogfood-fixes plan).
64
+ const FUZZY_MIN_MARKER_LEN = 6;
65
+ // "blooper" → "looker" is exactly this: 2 edits (drop the "b", substitute
66
+ // "p" for "k"). Found in the wild — see the guard test in blooper.test.ts.
67
+ const FUZZY_MAX_DISTANCE = 2;
68
+
69
+ interface MarkerMatch {
70
+ /** Normalized text of the transcript word that matched. */
71
+ surface: string;
72
+ /** False when this needed sound-alike/edit-distance rather than an exact hit. */
73
+ exact: boolean;
74
+ }
75
+
76
+ /**
77
+ * Whether a transcript word counts as the marker, and how.
78
+ *
79
+ * Exact match (today's rule) always wins first. Past that, the only fuzzy
80
+ * arm is a small edit distance — NOT `soundsSimilar` (§125,
81
+ * PHASE1-FINDINGS.md). The first field run of this feature paired the two
82
+ * arms as designed and got the worst of both: `soundsSimilar("builds",
83
+ * "blooper")` is true (shared "b" onset, score over its 0.34 floor) and cut
84
+ * 86.8% of a 125.9s video, while the pair `soundsSimilar` exists to catch —
85
+ * "looker" for "blooper" — is REJECTED by its own onset test (b/l differ)
86
+ * and only ever matched via Levenshtein anyway. Sound-alike was admitting
87
+ * garbage and catching nothing real, so it is gone; Levenshtein alone still
88
+ * catches "looker" (distance 2) and does not catch "builds" (distance 6).
89
+ */
90
+ function matchMarker(wordText: string, want: string): MarkerMatch | null {
91
+ const norm = normalizeToken(wordText);
92
+ if (!norm) return null;
93
+ if (norm === want) return { surface: norm, exact: true };
94
+ if (want.length < FUZZY_MIN_MARKER_LEN) return null;
95
+ if (levenshtein(norm, want) <= FUZZY_MAX_DISTANCE) {
96
+ return { surface: norm, exact: false };
97
+ }
98
+ return null;
41
99
  }
42
100
 
43
101
  /**
@@ -45,7 +103,9 @@ export interface BloopSpan {
45
103
  *
46
104
  * `marker` is matched with `normalizeToken`, the same normalizer the filler
47
105
  * detector uses, so case and trailing punctuation do not matter — ASR writes
48
- * the word as "blooper." with the period riding on it.
106
+ * the word as "blooper." with the period riding on it. Beyond exact text, a
107
+ * marker of at least `FUZZY_MIN_MARKER_LEN` characters also matches an ASR
108
+ * mishearing — see `matchMarker`.
49
109
  *
50
110
  * Returns spans in transcript order, non-overlapping.
51
111
  */
@@ -53,11 +113,15 @@ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpa
53
113
  const want = normalizeToken(marker);
54
114
  if (!want) return [];
55
115
  const words = transcript.words;
56
- const isMarker = (i: number): boolean => normalizeToken(words[i]?.text ?? "") === want;
116
+ const matchAt = (i: number): MarkerMatch | null => {
117
+ const w = words[i];
118
+ return w ? matchMarker(w.text, want) : null;
119
+ };
57
120
 
58
121
  const spans: BloopSpan[] = [];
59
122
  for (let i = 0; i < words.length; i++) {
60
- if (!isMarker(i)) continue;
123
+ const match = matchAt(i);
124
+ if (!match) continue;
61
125
 
62
126
  // Walk back over the attempt this marker spoiled, to the start of its
63
127
  // sentence. The marker's own text usually ENDS a sentence ("blooper."), so
@@ -70,10 +134,12 @@ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpa
70
134
  // one-word island of a sentence nobody finished.
71
135
  const prev = spans[spans.length - 1];
72
136
  let markers = 1;
137
+ let matched = match.exact ? [] : [match.surface];
73
138
  if (prev && start <= prev.endWord + 1) {
74
139
  spans.pop();
75
140
  start = prev.startWord;
76
141
  markers = prev.markers + 1;
142
+ matched = [...prev.matched, ...matched];
77
143
  }
78
144
 
79
145
  spans.push({
@@ -82,6 +148,8 @@ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpa
82
148
  startSec: words[start]!.start,
83
149
  endSec: words[i]!.end,
84
150
  markers,
151
+ marker: want,
152
+ matched,
85
153
  });
86
154
  }
87
155
  return spans;
@@ -98,5 +166,13 @@ export function formatBloopSpan(transcript: Transcript, span: BloopSpan): string
98
166
  .map((w) => w.text)
99
167
  .join(" ");
100
168
  const attempts = span.markers > 1 ? ` (${span.markers} attempts)` : "";
101
- return `"${said}"${attempts}`;
169
+ // A fuzzy hit must never be silent — this line is the safety net that
170
+ // makes on-by-default fuzzy matching acceptable (Task 3, editor-dogfood-fixes
171
+ // plan): a false positive shows up here instead of quietly cutting a good
172
+ // take.
173
+ const fuzzy =
174
+ span.matched.length > 0
175
+ ? " " + span.matched.map((m) => `matched "${m}" ~ "${span.marker}"`).join(", ")
176
+ : "";
177
+ return `"${said}"${attempts}${fuzzy}`;
102
178
  }
package/src/concat.ts ADDED
@@ -0,0 +1,461 @@
1
+ import { existsSync } from "node:fs";
2
+ import { readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises";
3
+ import { join } from "node:path";
4
+ import { z } from "zod/v4";
5
+ import { probe, type IngestTools } from "./ingest";
6
+ import { run } from "./exec";
7
+
8
+ /**
9
+ * `ossclip produce <folder>` (2026-08-05 field request, verbatim intent in
10
+ * .superpowers/sdd/folder-input-brief.md): a folder of camera-clip takes gets
11
+ * concatenated into ONE source before the normal produce pipeline ever sees
12
+ * it, instead of the user hand-concatenating with an agent first.
13
+ *
14
+ * Split per CLAUDE.md's pure/IO mandate — the bug that motivates the split is
15
+ * in the brief: a hand-built ffmpeg filtergraph corrupted by shell expansion
16
+ * (zsh's `:a` history modifier) silently played clip 1's audio in every
17
+ * concat slot, and nothing validated the filter STRING before it reached
18
+ * ffmpeg. `planFolderConcat`, `buildConcatFilter`, `folderManifestKey` and
19
+ * `assertAllClipsHaveAudio` are pure so each can be asserted on directly;
20
+ * `listFolderVideos` and `concatFolder` are the only I/O.
21
+ */
22
+
23
+ const VIDEO_EXTENSIONS = ["mov", "mp4", "m4v", "mkv", "webm", "avi"] as const;
24
+ const VIDEO_EXTENSION_SET = new Set<string>(VIDEO_EXTENSIONS);
25
+
26
+ function noVideoFilesError(folder: string): Error {
27
+ return new Error(
28
+ `no video files found directly inside ${folder} ` +
29
+ `(looked for: ${VIDEO_EXTENSIONS.map((e) => `.${e}`).join(", ")})`,
30
+ );
31
+ }
32
+
33
+ export interface ConcatEntry {
34
+ name: string;
35
+ mtimeMs: number;
36
+ size: number;
37
+ }
38
+
39
+ /**
40
+ * Order clips for concatenation. `name` (default, per the field request "sort
41
+ * them by name or date modified, name being default") is a PLAIN codepoint
42
+ * sort — comparing strings with `<`/`>` rather than `localeCompare`, which is
43
+ * what `ls` gives on a case-sensitive filesystem and a locale-aware sort would
44
+ * NOT: it reorders case and punctuation differently per machine locale, which
45
+ * would make the same folder concat in a different order on a different
46
+ * machine. `mtime` ties (a batch copy that preserved one timestamp across
47
+ * several files) fall back to name so the order stays reproducible either way.
48
+ */
49
+ export function planFolderConcat(
50
+ entries: readonly ConcatEntry[],
51
+ sort: "name" | "mtime",
52
+ ): string[] {
53
+ const byName = (a: ConcatEntry, b: ConcatEntry): number =>
54
+ a.name < b.name ? -1 : a.name > b.name ? 1 : 0;
55
+ const cmp = sort === "name" ? byName : (a: ConcatEntry, b: ConcatEntry) => a.mtimeMs - b.mtimeMs || byName(a, b);
56
+ return [...entries].sort(cmp).map((e) => e.name);
57
+ }
58
+
59
+ /**
60
+ * A deterministic, order-independent identity for a folder's clip set —
61
+ * sorted by name (a canonical order regardless of `readdir`'s OS-dependent
62
+ * enumeration order) before joining, so the same files always hash the same
63
+ * way. `sort` is folded in because a `--sort` flip changes the concat's
64
+ * actual bytes (different clip order), not just how it was chosen.
65
+ *
66
+ * Fix for a review finding on the first cut of this feature: the workdir
67
+ * hash used to be derived from the FOLDER PATH alone, which is stable across
68
+ * content changes — but `audio.wav`, `transcript.json`, the content-rect
69
+ * cache and the mezzanine are all existence-keyed inside that same workdir.
70
+ * Adding a take (or flipping --sort) rebuilt `source-concat.mp4` correctly
71
+ * but silently reused every one of those, producing a video with captions
72
+ * transcribed against the PREVIOUS concat. Hashing the manifest content here
73
+ * — the same invariant a file input already has via `sha1File` — means a
74
+ * changed folder gets a fresh workdir, and every derived cache is fresh too.
75
+ *
76
+ * Serialized with JSON.stringify, not a `:`/`|` delimiter join (audit fix):
77
+ * a filename is user-controlled free text that can itself contain the
78
+ * delimiters, letting two DIFFERENT entry sets serialize to one identical
79
+ * key — `a:1` sized 2 and `a` sized `1:2` collide under a `:` join, and a
80
+ * collision here means one folder silently reuses another's transcript and
81
+ * mezzanine. JSON escapes the filename instead of trusting it. This changed
82
+ * every existing folder workdir hash once — a one-time cache invalidation
83
+ * (fresh workdir, full re-concat/re-transcribe on the next run), accepted as
84
+ * the cost of an injection-proof key.
85
+ */
86
+ export function folderManifestKey(entries: readonly ConcatEntry[], sort: "name" | "mtime"): string {
87
+ const canonical = [...entries]
88
+ .sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0))
89
+ .map((e) => ({ name: e.name, size: e.size, mtimeMs: e.mtimeMs }));
90
+ return JSON.stringify({ sort, entries: canonical });
91
+ }
92
+
93
+ /**
94
+ * The `-filter_complex` string concatenating `n` inputs into one output.
95
+ * Each input gets its OWN scale+pad+fps+setsar+format chain — letterboxed to
96
+ * `target`, never cropped, since produce's own framing decides crops later —
97
+ * and its own audio resample, before the `concat` filter joins them.
98
+ *
99
+ * Rotation: ffmpeg auto-rotates on decode by default (R27 §119, `probe()` in
100
+ * ingest.ts relies on the same fact), so every `[i:v]` input here is ASSUMED
101
+ * to already arrive in DISPLAYED orientation — the scale/pad math needs no
102
+ * separate rotation step. This is an assumption carried over from R27 §119,
103
+ * not something the folder-input verification run could independently
104
+ * confirm: a letterboxed, correctly-proportioned 1080x1920 output is also
105
+ * what a WRONG rotation assumption would produce once padded to a portrait
106
+ * canvas, so that run couldn't distinguish "handled correctly" from
107
+ * "accidentally looks fine." Flagged rather than overclaimed per CLAUDE.md.
108
+ *
109
+ * `n` labels of each kind, never more or fewer, and the tail's `[vI][aI]`
110
+ * pairs are built in the SAME loop that emits them — this is the direct
111
+ * regression test target for the field bug (see module comment): a
112
+ * hand-built graph had a slot silently reference input 0's audio a second
113
+ * time instead of its own index, and nothing caught the STRING being wrong.
114
+ */
115
+ export function buildConcatFilter(n: number, target: { w: number; h: number }): string {
116
+ const chains: string[] = [];
117
+ const tail: string[] = [];
118
+ for (let i = 0; i < n; i++) {
119
+ chains.push(
120
+ `[${i}:v]scale=${target.w}:${target.h}:force_original_aspect_ratio=decrease,` +
121
+ `pad=${target.w}:${target.h}:(ow-iw)/2:(oh-ih)/2,fps=30,setsar=1,format=yuv420p[v${i}]`,
122
+ );
123
+ chains.push(`[${i}:a]aresample=48000,aformat=channel_layouts=stereo[a${i}]`);
124
+ tail.push(`[v${i}][a${i}]`);
125
+ }
126
+ return `${chains.join(";")};${tail.join("")}concat=n=${n}:v=1:a=1[outv][outa]`;
127
+ }
128
+
129
+ /**
130
+ * `buildConcatFilter` emits `[i:a]` unconditionally for every input — it has
131
+ * no way to know a clip is silent. Handed a video-only clip (b-roll with no
132
+ * audio stream), ffmpeg dies deep inside the filtergraph with a bare stream-
133
+ * specifier error that names neither the clip nor the reason. `probe()`
134
+ * already reports `hasAudio`; failing BEFORE ffmpeg ever runs keeps faith
135
+ * with the brief's "a file that probe() rejects is an error naming the file,
136
+ * not a silent skip" — a clip with no audio is the same class of problem,
137
+ * just discovered an instant later than "no video stream at all".
138
+ *
139
+ * `evaluateAudioProbes` is the guard's pure decision (0.1.9 first-contact,
140
+ * 2026-08-05): the fail-fast pool below fires it the moment a silent clip is
141
+ * seen, when only SOME probes have settled — so the verdict carries both the
142
+ * offenders known at that moment and how many clips were never checked, and
143
+ * the message can be honest about the difference instead of implying a
144
+ * complete list it doesn't have.
145
+ */
146
+ export interface AudioGuardFailure {
147
+ /** Clips known silent when the guard fired, in enumeration order. */
148
+ offenders: string[];
149
+ /** Clips whose probes had not settled when the guard fired. */
150
+ uncheckedCount: number;
151
+ }
152
+
153
+ export function evaluateAudioProbes(
154
+ settled: ReadonlyArray<{ name: string; hasAudio: boolean }>,
155
+ total: number,
156
+ ): AudioGuardFailure | null {
157
+ const offenders = settled.filter((c) => !c.hasAudio).map((c) => c.name);
158
+ if (offenders.length === 0) return null;
159
+ return { offenders, uncheckedCount: total - settled.length };
160
+ }
161
+
162
+ export function audioGuardMessage(failure: AudioGuardFailure): string {
163
+ const { offenders, uncheckedCount } = failure;
164
+ // "Result", not "clip": probes settle in parallel, so the unchecked set can
165
+ // include clips EARLIER in concat order than the offender — "later clips"
166
+ // would overclaim an ordering the pool doesn't have (review, Minor 1).
167
+ const stopped =
168
+ uncheckedCount > 0
169
+ ? ` Stopped at the first missing-audio result; ${uncheckedCount} ` +
170
+ `clip${uncheckedCount === 1 ? " was" : "s were"} not checked.`
171
+ : "";
172
+ // The ~/Downloads sentence is the 0.1.9 first-contact lesson: the guard
173
+ // fired CORRECTLY, but on the wrong folder — the wizard had been answered
174
+ // with all of ~/Downloads, and "13 clips lack audio" never suggested the
175
+ // user was one directory too high.
176
+ return (
177
+ `no audio stream in: ${offenders.join(", ")} — produce cuts by silence, so ` +
178
+ "every clip in a folder concat needs one (a silent b-roll clip can't be " +
179
+ "concatenated this way)." +
180
+ stopped +
181
+ " If this is a mixed folder like ~/Downloads rather than a folder of just " +
182
+ "the takes to concatenate, point produce at the clips folder instead."
183
+ );
184
+ }
185
+
186
+ export function assertAllClipsHaveAudio(
187
+ clips: ReadonlyArray<{ name: string; hasAudio: boolean }>,
188
+ ): void {
189
+ const failure = evaluateAudioProbes(clips, clips.length);
190
+ if (failure !== null) throw new Error(audioGuardMessage(failure));
191
+ }
192
+
193
+ /**
194
+ * Bounded, fail-fast probe pool (0.1.9 first-contact, 2026-08-05): the guard
195
+ * used to `Promise.all` every probe — pointed at ~/Downloads, that meant
196
+ * 4m32s of ffprobe over ~100 unrelated videos before the error the FIRST
197
+ * silent clip already implied. Two fixes in one shape: a concurrency bound so
198
+ * a big folder doesn't stampede the disk with a hundred simultaneous
199
+ * ffprobes, and an immediate reject on the first missing-audio result — the
200
+ * verdict is built from whatever has settled at that moment, and in-flight
201
+ * probes are simply ignored (ffprobe is a short-lived read-only process;
202
+ * letting it finish unobserved is harmless).
203
+ *
204
+ * Generic over `probeOne` so the scheduling is testable without ffprobe; the
205
+ * abort/offenders DECISION lives in `evaluateAudioProbes` (pure), this is the
206
+ * IO glue around it.
207
+ */
208
+ export async function probeClipsWithAudioGuard<T extends { hasAudio: boolean }>(
209
+ names: readonly string[],
210
+ probeOne: (name: string, index: number) => Promise<T>,
211
+ concurrency = 8,
212
+ ): Promise<T[]> {
213
+ if (names.length === 0) return [];
214
+ const results: (T | undefined)[] = new Array(names.length);
215
+ return await new Promise<T[]>((resolveAll, rejectAll) => {
216
+ let nextIndex = 0;
217
+ let settledCount = 0;
218
+ let finished = false;
219
+
220
+ const launch = (): void => {
221
+ if (finished || nextIndex >= names.length) return;
222
+ const i = nextIndex;
223
+ nextIndex++;
224
+ const name = names[i];
225
+ if (name === undefined) return; // unreachable: i < names.length
226
+ // Deferred through a resolved promise so a probeOne that throws
227
+ // SYNCHRONOUSLY rejects the pool instead of deadlocking it (review,
228
+ // Minor 2): a sync throw out of a replacement `launch()` inside the
229
+ // success handler below would otherwise escape both handlers, leaving
230
+ // the pool one settle short of ever resolving. Unreachable with the
231
+ // real async probe, but the generic API must not depend on that.
232
+ Promise.resolve()
233
+ .then(() => probeOne(name, i))
234
+ .then(
235
+ (result) => {
236
+ if (finished) return;
237
+ results[i] = result;
238
+ settledCount++;
239
+ if (!result.hasAudio) {
240
+ const settled: Array<{ name: string; hasAudio: boolean }> = [];
241
+ for (let j = 0; j < names.length; j++) {
242
+ const r = results[j];
243
+ const n = names[j];
244
+ if (r !== undefined && n !== undefined) settled.push({ name: n, hasAudio: r.hasAudio });
245
+ }
246
+ const failure = evaluateAudioProbes(settled, names.length);
247
+ // Never null here — the result that brought us into this branch
248
+ // lacks audio — but throwing through the pure function keeps one
249
+ // single source of truth for the verdict.
250
+ if (failure !== null) {
251
+ finished = true;
252
+ rejectAll(new Error(audioGuardMessage(failure)));
253
+ }
254
+ return;
255
+ }
256
+ if (settledCount === names.length) {
257
+ finished = true;
258
+ resolveAll(results as T[]);
259
+ } else {
260
+ launch();
261
+ }
262
+ },
263
+ (err: unknown) => {
264
+ // A probe that FAILS (e.g. `no video stream in <path>`) propagates
265
+ // as-is, same as the Promise.all it replaced — an error naming the
266
+ // file, not a silent skip (folder-input-brief.md).
267
+ if (finished) return;
268
+ finished = true;
269
+ rejectAll(err instanceof Error ? err : new Error(String(err)));
270
+ },
271
+ );
272
+ };
273
+ const initial = Math.min(concurrency, names.length);
274
+ for (let k = 0; k < initial; k++) launch();
275
+ });
276
+ }
277
+
278
+ const ConcatManifestSchema = z.object({
279
+ sort: z.enum(["name", "mtime"]),
280
+ entries: z.array(
281
+ z.object({
282
+ name: z.string(),
283
+ mtimeMs: z.number(),
284
+ size: z.number(),
285
+ durationSec: z.number(),
286
+ }),
287
+ ),
288
+ });
289
+ type ConcatManifest = z.infer<typeof ConcatManifestSchema>;
290
+
291
+ /**
292
+ * The cached build is reusable only if EVERY current file is present in the
293
+ * manifest with the same size and mtime (a changed byte count or timestamp
294
+ * means the clip could have been re-exported), the counts match (a clip
295
+ * removed leaves no trace otherwise), and the sort mode is the one the
296
+ * manifest was built under (a `--sort` change reorders the clips, so a
297
+ * same-files cache is still the WRONG concat).
298
+ *
299
+ * Belt-and-suspenders alongside `folderManifestKey`: the workdir is now
300
+ * content-addressed too, so in practice a stale manifest can only be reached
301
+ * by a hash collision or a folder mutated mid-run — this is what catches
302
+ * either without trusting the hash alone.
303
+ */
304
+ function manifestStillValid(manifest: ConcatManifest, sort: "name" | "mtime", current: readonly ConcatEntry[]): boolean {
305
+ if (manifest.sort !== sort) return false;
306
+ if (manifest.entries.length !== current.length) return false;
307
+ const byName = new Map(manifest.entries.map((e) => [e.name, e]));
308
+ return current.every((e) => {
309
+ const prev = byName.get(e.name);
310
+ return prev !== undefined && prev.mtimeMs === e.mtimeMs && prev.size === e.size;
311
+ });
312
+ }
313
+
314
+ export interface FolderListing {
315
+ entries: ConcatEntry[];
316
+ /** Files skipped for not matching a video extension (dotfiles excluded). */
317
+ nonVideoCount: number;
318
+ }
319
+
320
+ /**
321
+ * Enumerate the video files directly inside `folder` (no recursion — a
322
+ * subfolder is out of scope, not "ignored", so it is never counted).
323
+ *
324
+ * Symlinks to a regular file are followed (`stat`, which resolves the link)
325
+ * rather than dropped — a folder of symlinks into another drive is a normal
326
+ * way to stage takes, and silently enumerating zero clips from it would be a
327
+ * worse surprise than the extra `stat` call. A broken symlink or a symlink to
328
+ * a directory stats as "not a file" and is skipped without counting, the same
329
+ * as a real subfolder. Dotfiles (`.DS_Store` and friends) are skipped
330
+ * entirely and never counted — they are not a folder content decision the
331
+ * user made, so reporting them as "non-video files ignored" would be noise.
332
+ */
333
+ export async function listFolderVideos(folder: string): Promise<FolderListing> {
334
+ const dirents = await readdir(folder, { withFileTypes: true });
335
+ let nonVideoCount = 0;
336
+ const entries: ConcatEntry[] = [];
337
+ for (const d of dirents) {
338
+ if (d.name.startsWith(".")) continue;
339
+ let isFile = d.isFile();
340
+ if (!isFile && d.isSymbolicLink()) {
341
+ try {
342
+ isFile = (await stat(join(folder, d.name))).isFile();
343
+ } catch {
344
+ isFile = false; // broken symlink — treated like a subfolder: skipped, not counted
345
+ }
346
+ }
347
+ if (!isFile) continue; // a real subfolder, or a symlink to one
348
+ const dot = d.name.lastIndexOf(".");
349
+ const ext = dot >= 0 ? d.name.slice(dot + 1).toLowerCase() : "";
350
+ if (!VIDEO_EXTENSION_SET.has(ext)) {
351
+ nonVideoCount++;
352
+ continue;
353
+ }
354
+ const st = await stat(join(folder, d.name));
355
+ entries.push({ name: d.name, mtimeMs: st.mtimeMs, size: st.size });
356
+ }
357
+ if (entries.length === 0) throw noVideoFilesError(folder);
358
+ return { entries, nonVideoCount };
359
+ }
360
+
361
+ export interface FolderConcatResult {
362
+ /** The intermediate file — hand this to the rest of produce. */
363
+ path: string;
364
+ /** In final concat order, for the "one line per clip" console report. */
365
+ clips: Array<{ name: string; durationSec: number }>;
366
+ /** Files skipped for not matching a video extension. */
367
+ nonVideoCount: number;
368
+ /** True when the existing `source-concat.mp4` was reused, not rebuilt. */
369
+ cached: boolean;
370
+ /** The concat's own total duration (ffprobe'd from the output). */
371
+ durationSec: number;
372
+ }
373
+
374
+ /**
375
+ * Order and concat `listing`'s clips into `<workDir>/source-concat.mp4`,
376
+ * caching on a manifest of names+sizes+mtimes so an unchanged folder skips
377
+ * the re-encode. `listing` comes from `listFolderVideos` — the caller
378
+ * enumerates once, up front, because it ALSO needs the listing to derive the
379
+ * workdir's content-addressed hash (`folderManifestKey`) before this can
380
+ * even be called with a `workDir` to write into.
381
+ */
382
+ export async function concatFolder(
383
+ tools: IngestTools,
384
+ folder: string,
385
+ listing: FolderListing,
386
+ workDir: string,
387
+ sort: "name" | "mtime",
388
+ target: { w: number; h: number },
389
+ ): Promise<FolderConcatResult> {
390
+ const { entries: current, nonVideoCount } = listing;
391
+ if (current.length === 0) throw noVideoFilesError(folder);
392
+ const order = planFolderConcat(current, sort);
393
+
394
+ const outPath = join(workDir, "source-concat.mp4");
395
+ const manifestPath = join(workDir, "source-concat.json");
396
+ if (existsSync(outPath) && existsSync(manifestPath)) {
397
+ const parsed = ConcatManifestSchema.safeParse(JSON.parse(await readFile(manifestPath, "utf8")));
398
+ if (parsed.success && manifestStillValid(parsed.data, sort, current)) {
399
+ const byName = new Map(parsed.data.entries.map((e) => [e.name, e]));
400
+ const outProbe = await probe(tools, outPath);
401
+ return {
402
+ path: outPath,
403
+ clips: order.map((name) => ({ name, durationSec: byName.get(name)!.durationSec })),
404
+ nonVideoCount,
405
+ cached: true,
406
+ durationSec: outProbe.duration,
407
+ };
408
+ }
409
+ }
410
+
411
+ // A file with a video EXTENSION that fails to probe (no video stream) is an
412
+ // error naming the file, not a silent skip (folder-input-brief.md) — so
413
+ // `probe()`'s own "no video stream in <path>" is left to propagate rather
414
+ // than caught here. The pool bounds concurrency and rejects on the FIRST
415
+ // silent clip instead of probing the whole folder before refusing (0.1.9
416
+ // first-contact, 2026-08-05 — see probeClipsWithAudioGuard).
417
+ const probes = await probeClipsWithAudioGuard(order, (name) => probe(tools, join(folder, name)));
418
+
419
+ const filter = buildConcatFilter(order.length, target);
420
+ const inputArgs = order.flatMap((name) => ["-i", join(folder, name)]);
421
+ // Encode to a sibling temp path, rename only on success — same reasoning as
422
+ // `bakeNormalizedSource` in normalize.ts (R27 §125): ffmpeg writes the
423
+ // container header as it goes, so a bake that dies mid-graph leaves a file
424
+ // with no `moov` atom, and a cache keyed on EXISTENCE would reuse that
425
+ // corpse forever. Rename is atomic on a POSIX filesystem.
426
+ const partial = `${outPath}.partial.mp4`;
427
+ try {
428
+ await run(tools.ffmpegPath, [
429
+ "-y",
430
+ ...inputArgs,
431
+ "-filter_complex", filter,
432
+ "-map", "[outv]",
433
+ "-map", "[outa]",
434
+ "-c:v", "libx264", "-preset", "medium", "-crf", "18",
435
+ "-c:a", "aac", "-b:a", "192k",
436
+ partial,
437
+ ]);
438
+ await rename(partial, outPath);
439
+ } catch (err) {
440
+ await rm(partial, { force: true });
441
+ throw err;
442
+ }
443
+
444
+ const manifest: ConcatManifest = {
445
+ sort,
446
+ entries: current.map((e) => ({
447
+ ...e,
448
+ durationSec: probes[order.indexOf(e.name)]!.duration,
449
+ })),
450
+ };
451
+ await writeFile(manifestPath, JSON.stringify(manifest, null, 2));
452
+
453
+ const outProbe = await probe(tools, outPath);
454
+ return {
455
+ path: outPath,
456
+ clips: order.map((name, i) => ({ name, durationSec: probes[i]!.duration })),
457
+ nonVideoCount,
458
+ cached: false,
459
+ durationSec: outProbe.duration,
460
+ };
461
+ }