@ossclip/core 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +27 -0
- package/README.md +20 -0
- package/package.json +29 -0
- package/src/analyze.ts +299 -0
- package/src/assemble.ts +124 -0
- package/src/browser.ts +24 -0
- package/src/captions.ts +92 -0
- package/src/clip.ts +306 -0
- package/src/config.ts +66 -0
- package/src/content-rect-detect.ts +162 -0
- package/src/content-rect.ts +324 -0
- package/src/cover.ts +216 -0
- package/src/cta.ts +68 -0
- package/src/cutlist.ts +170 -0
- package/src/exec.ts +36 -0
- package/src/face.ts +519 -0
- package/src/fill.ts +110 -0
- package/src/framing.ts +277 -0
- package/src/grounding.ts +130 -0
- package/src/index.ts +27 -0
- package/src/ingest.ts +83 -0
- package/src/normalize.ts +397 -0
- package/src/overrides.ts +509 -0
- package/src/phonetics.ts +129 -0
- package/src/producer/anthropic.ts +73 -0
- package/src/producer/beats.ts +330 -0
- package/src/producer/claude-cli.ts +150 -0
- package/src/producer/gemini.ts +197 -0
- package/src/producer/index.ts +217 -0
- package/src/producer/mock.ts +101 -0
- package/src/producer/provider.ts +42 -0
- package/src/producer/repair.ts +474 -0
- package/src/producer/scene-props.ts +212 -0
- package/src/producer/tiered.ts +56 -0
- package/src/producer/usage.ts +426 -0
- package/src/report.ts +36 -0
- package/src/scene-registry.ts +246 -0
- package/src/scene-schema.ts +203 -0
- package/src/schema.ts +177 -0
- package/src/source-text.ts +348 -0
- package/src/timemap.ts +115 -0
- package/src/transcribe.ts +67 -0
- package/src/zoom.ts +154 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Muhammad Ahsan Ayaz
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
22
|
+
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
Note (not part of the MIT grant above, which covers this repository's own
|
|
26
|
+
code): rendering depends on Remotion, which is source-available under its own
|
|
27
|
+
two-tier licence — see https://github.com/remotion-dev/remotion/blob/main/LICENSE.md.
|
package/README.md
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# @ossclip/core
|
|
2
|
+
|
|
3
|
+
Part of [**ossclip**](https://github.com/AhsanAyaz/ossclip) — a local-first CLI that turns a talking-head take into a finished short: silence and filler words cut, word-timed kinetic captions, face-aware framing, and LLM-planned code-rendered on-screen graphics.
|
|
4
|
+
|
|
5
|
+
This package is the framework-free pipeline: schema, transcription, analysis, cutlist, captions, framing, and the LLM producer.
|
|
6
|
+
|
|
7
|
+
It is published so the CLI can depend on it and so the pieces are reusable, but the supported entry point is the CLI:
|
|
8
|
+
|
|
9
|
+
```sh
|
|
10
|
+
npm install -g ossclip
|
|
11
|
+
ossclip doctor
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
APIs here move between rounds — pin an exact version if you depend on them directly.
|
|
15
|
+
|
|
16
|
+
**Documentation:** [github.com/AhsanAyaz/ossclip](https://github.com/AhsanAyaz/ossclip)
|
|
17
|
+
|
|
18
|
+
## Licence
|
|
19
|
+
|
|
20
|
+
MIT. Rendering depends on [Remotion](https://www.remotion.dev/), which carries [its own two-tier licence](https://github.com/remotion-dev/remotion/blob/main/LICENSE.md).
|
package/package.json
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@ossclip/core",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "ossclip's framework-free pipeline: schema, transcription, analysis, cutlist, captions, framing, and the LLM producer",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"license": "MIT",
|
|
7
|
+
"repository": {
|
|
8
|
+
"type": "git",
|
|
9
|
+
"url": "git+https://github.com/AhsanAyaz/ossclip.git",
|
|
10
|
+
"directory": "packages/core"
|
|
11
|
+
},
|
|
12
|
+
"exports": {
|
|
13
|
+
".": "./src/index.ts",
|
|
14
|
+
"./browser": "./src/browser.ts"
|
|
15
|
+
},
|
|
16
|
+
"files": [
|
|
17
|
+
"README.md",
|
|
18
|
+
"src"
|
|
19
|
+
],
|
|
20
|
+
"dependencies": {
|
|
21
|
+
"@anthropic-ai/sdk": "^0.115.0",
|
|
22
|
+
"zod": "^3.25.0"
|
|
23
|
+
},
|
|
24
|
+
"homepage": "https://github.com/AhsanAyaz/ossclip#readme",
|
|
25
|
+
"bugs": {
|
|
26
|
+
"url": "https://github.com/AhsanAyaz/ossclip/issues"
|
|
27
|
+
},
|
|
28
|
+
"author": "Muhammad Ahsan Ayaz"
|
|
29
|
+
}
|
package/src/analyze.ts
ADDED
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
import { run } from "./exec";
|
|
2
|
+
import type { Analysis, Span, Transcript } from "./schema";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Standalone interjections only. Deliberately excludes "like"/"you know"/"ah"/"oh":
|
|
6
|
+
* without an LLM adjudicating, false positives feel far worse than fillers.
|
|
7
|
+
*/
|
|
8
|
+
const FILLER_WORDS = new Set(["um", "uh", "uhm", "erm", "er", "hmm", "hm", "mmm", "mm", "mhm", "uh-huh"]);
|
|
9
|
+
|
|
10
|
+
export function normalizeToken(text: string): string {
|
|
11
|
+
return text
|
|
12
|
+
.toLowerCase()
|
|
13
|
+
.replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}-]+$/gu, "");
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
export interface SilenceDetectOptions {
|
|
17
|
+
ffmpegPath: string;
|
|
18
|
+
noiseDb?: number;
|
|
19
|
+
minDuration?: number;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export interface LevelStats {
|
|
23
|
+
/** Noise floor — 10th percentile of per-window RMS, dBFS. */
|
|
24
|
+
floorDb: number;
|
|
25
|
+
/** Speech level — 90th percentile of per-window RMS, dBFS. */
|
|
26
|
+
speechDb: number;
|
|
27
|
+
/** Silence threshold derived from the two, dBFS. */
|
|
28
|
+
thresholdDb: number;
|
|
29
|
+
/** Per-window RMS in dBFS, in order. */
|
|
30
|
+
windowsDb: number[];
|
|
31
|
+
/** Window length, seconds. */
|
|
32
|
+
windowSec: number;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** Window length used for level measurement, seconds. */
|
|
36
|
+
const WINDOW_SEC = 0.1;
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Typical level of a dB series — the MEDIAN window, not an energy average.
|
|
40
|
+
* Energy averaging is peak-dominated: one speech window straddling the edge of
|
|
41
|
+
* a 3 s silence drags its measured level from −51 dB to −26 dB and the region
|
|
42
|
+
* stops looking like dead air. The median ignores that single window.
|
|
43
|
+
*/
|
|
44
|
+
export function typicalLevelDb(windowsDb: readonly number[]): number {
|
|
45
|
+
if (windowsDb.length === 0) return Number.NEGATIVE_INFINITY;
|
|
46
|
+
const sorted = [...windowsDb].sort((a, b) => a - b);
|
|
47
|
+
return percentile(sorted, 0.5);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** Typical level of [start, end) from a per-window dB series. */
|
|
51
|
+
export function regionLevelDb(
|
|
52
|
+
windowsDb: readonly number[],
|
|
53
|
+
windowSec: number,
|
|
54
|
+
start: number,
|
|
55
|
+
end: number,
|
|
56
|
+
): number {
|
|
57
|
+
const lo = Math.max(0, Math.floor(start / windowSec));
|
|
58
|
+
const hi = Math.min(windowsDb.length, Math.ceil(end / windowSec));
|
|
59
|
+
return typicalLevelDb(windowsDb.slice(lo, hi));
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** Widest and narrowest silence thresholds we'll ever derive, dBFS. */
|
|
63
|
+
const THRESHOLD_FLOOR = -40;
|
|
64
|
+
const THRESHOLD_CEIL = -20;
|
|
65
|
+
/**
|
|
66
|
+
* How far below the speech level the threshold sits.
|
|
67
|
+
*
|
|
68
|
+
* Anchored to speech, NOT to the noise floor: a "silent" room is only quiet on
|
|
69
|
+
* average. Measured room tone in real footage averaged −49 dB but peaked at
|
|
70
|
+
* −28 dB, and silencedetect needs a *continuous* run below the threshold — so
|
|
71
|
+
* anything under about −27 dB broke a 3 s pause into 0.4 s fragments and the
|
|
72
|
+
* pause was never cut. The mean tells you nothing; the peaks decide.
|
|
73
|
+
*/
|
|
74
|
+
const SPEECH_DROP = 12;
|
|
75
|
+
/** Never put the threshold within this much of the measured noise floor… */
|
|
76
|
+
const FLOOR_HEADROOM = 6;
|
|
77
|
+
/** …nor this close to the speech level, or speech itself reads as silence. */
|
|
78
|
+
const SPEECH_HEADROOM = 8;
|
|
79
|
+
|
|
80
|
+
export function deriveThreshold(floorDb: number, speechDb: number): number {
|
|
81
|
+
const bounded = Math.min(Math.max(speechDb - SPEECH_DROP, THRESHOLD_FLOOR), THRESHOLD_CEIL);
|
|
82
|
+
const lo = floorDb + FLOOR_HEADROOM;
|
|
83
|
+
const hi = speechDb - SPEECH_HEADROOM;
|
|
84
|
+
// Material with no dynamic range at all (near-silent or clipped throughout)
|
|
85
|
+
// leaves no valid window — the midpoint is the least-wrong answer.
|
|
86
|
+
if (lo > hi) return (floorDb + speechDb) / 2;
|
|
87
|
+
return Math.min(Math.max(bounded, lo), hi);
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export function percentile(sorted: readonly number[], p: number): number {
|
|
91
|
+
if (sorted.length === 0) return Number.NaN;
|
|
92
|
+
return sorted[Math.min(sorted.length - 1, Math.max(0, Math.floor(p * (sorted.length - 1))))]!;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Measure the take's own noise floor and speech level, so the silence
|
|
97
|
+
* threshold tracks the recording instead of a hardcoded dB value. A hot
|
|
98
|
+
* lav mic and a quiet room condenser have floors 25 dB apart; one fixed
|
|
99
|
+
* threshold silently no-ops on the loud one.
|
|
100
|
+
*/
|
|
101
|
+
export async function measureLevels(opts: { ffmpegPath: string }, audioPath: string): Promise<LevelStats> {
|
|
102
|
+
// 1600 samples @ 16 kHz (the ASR wav rate) = 100 ms windows.
|
|
103
|
+
const { stdout, stderr } = await run(
|
|
104
|
+
opts.ffmpegPath,
|
|
105
|
+
[
|
|
106
|
+
"-hide_banner", "-nostats", "-i", audioPath,
|
|
107
|
+
"-af",
|
|
108
|
+
"asetnsamples=n=1600,astats=metadata=1:reset=1,ametadata=print:key=lavfi.astats.Overall.RMS_level:file=-",
|
|
109
|
+
"-f", "null", "-",
|
|
110
|
+
],
|
|
111
|
+
{ allowNonZero: true },
|
|
112
|
+
);
|
|
113
|
+
const values: number[] = [];
|
|
114
|
+
for (const line of `${stdout}\n${stderr}`.split("\n")) {
|
|
115
|
+
const m = line.match(/lavfi\.astats\.Overall\.RMS_level=(-?[\d.]+|-?inf)/);
|
|
116
|
+
if (!m) continue;
|
|
117
|
+
const v = Number(m[1]);
|
|
118
|
+
if (Number.isFinite(v)) values.push(v);
|
|
119
|
+
}
|
|
120
|
+
if (values.length === 0) {
|
|
121
|
+
// No usable measurement — fall back to the old fixed threshold.
|
|
122
|
+
return { floorDb: -60, speechDb: -20, thresholdDb: -35, windowsDb: [], windowSec: WINDOW_SEC };
|
|
123
|
+
}
|
|
124
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
125
|
+
const floorDb = percentile(sorted, 0.1);
|
|
126
|
+
const speechDb = percentile(sorted, 0.9);
|
|
127
|
+
return {
|
|
128
|
+
floorDb,
|
|
129
|
+
speechDb,
|
|
130
|
+
thresholdDb: deriveThreshold(floorDb, speechDb),
|
|
131
|
+
windowsDb: values,
|
|
132
|
+
windowSec: WINDOW_SEC,
|
|
133
|
+
};
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/** Acoustic silences via ffmpeg silencedetect (parsed from stderr). */
|
|
137
|
+
export async function detectSilences(opts: SilenceDetectOptions, audioPath: string): Promise<Span[]> {
|
|
138
|
+
const noise = opts.noiseDb ?? (await measureLevels(opts, audioPath)).thresholdDb;
|
|
139
|
+
const minDur = opts.minDuration ?? 0.35;
|
|
140
|
+
const { stderr } = await run(
|
|
141
|
+
opts.ffmpegPath,
|
|
142
|
+
["-i", audioPath, "-af", `silencedetect=noise=${noise}dB:d=${minDur}`, "-f", "null", "-"],
|
|
143
|
+
{ allowNonZero: true },
|
|
144
|
+
);
|
|
145
|
+
const silences: Span[] = [];
|
|
146
|
+
let pendingStart: number | null = null;
|
|
147
|
+
for (const line of stderr.split("\n")) {
|
|
148
|
+
const startMatch = line.match(/silence_start:\s*(-?[\d.]+)/);
|
|
149
|
+
if (startMatch) pendingStart = Math.max(0, Number(startMatch[1]));
|
|
150
|
+
const endMatch = line.match(/silence_end:\s*(-?[\d.]+)/);
|
|
151
|
+
if (endMatch && pendingStart !== null) {
|
|
152
|
+
silences.push({ start: pendingStart, end: Number(endMatch[1]) });
|
|
153
|
+
pendingStart = null;
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
if (pendingStart !== null) silences.push({ start: pendingStart, end: Number.POSITIVE_INFINITY });
|
|
157
|
+
return silences;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
/** A cuttable region shorter than this isn't worth considering. */
|
|
161
|
+
const MIN_CUTTABLE = 0.1;
|
|
162
|
+
/**
|
|
163
|
+
* Mean level this far below speech is dead air, whatever the transcript claims.
|
|
164
|
+
* Quiet or off-mic speech still averages nearer the speech level than this.
|
|
165
|
+
*/
|
|
166
|
+
const DEAD_AIR_DROP = 25;
|
|
167
|
+
|
|
168
|
+
/** Subtract `blockers` from `span`, returning the surviving pieces. */
|
|
169
|
+
export function subtractSpans(span: Span, blockers: readonly Span[]): Span[] {
|
|
170
|
+
let pieces: Span[] = [{ ...span }];
|
|
171
|
+
for (const b of blockers) {
|
|
172
|
+
const next: Span[] = [];
|
|
173
|
+
for (const p of pieces) {
|
|
174
|
+
if (b.end <= p.start || b.start >= p.end) {
|
|
175
|
+
next.push(p);
|
|
176
|
+
continue;
|
|
177
|
+
}
|
|
178
|
+
if (b.start > p.start) next.push({ start: p.start, end: b.start });
|
|
179
|
+
if (b.end < p.end) next.push({ start: b.end, end: p.end });
|
|
180
|
+
}
|
|
181
|
+
pieces = next;
|
|
182
|
+
}
|
|
183
|
+
return pieces;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* Acoustics decide what is cuttable; the transcript only vetoes.
|
|
188
|
+
*
|
|
189
|
+
* The previous rule ("cut only where a transcript gap and a silence agree")
|
|
190
|
+
* cannot fire on real whisper output: with `-ml 1` the word stamps are
|
|
191
|
+
* contiguous — each word's end IS the next word's start — so pauses are
|
|
192
|
+
* absorbed into word durations and the transcript reports no gap to agree
|
|
193
|
+
* with. Measured on a real 68 s take: 164/167 word boundaries contiguous,
|
|
194
|
+
* every detected silence landing *inside* a word, zero agreed pauses, zero
|
|
195
|
+
* cuts.
|
|
196
|
+
*
|
|
197
|
+
* Inverted rule: a region below the silence threshold contains no audible
|
|
198
|
+
* speech by definition, so it is cuttable. The transcript vetoes only the
|
|
199
|
+
* pathological case where a whole non-filler word is claimed to live inside
|
|
200
|
+
* that silence — a real signal conflict, where the safe move is to keep it.
|
|
201
|
+
*
|
|
202
|
+
* That veto is itself overridden when the region's MEAN energy is far below
|
|
203
|
+
* the speech level (`levels` supplied): whisper stamps words over dead air
|
|
204
|
+
* often enough that trusting it there would cancel obvious lead-in trims,
|
|
205
|
+
* while genuinely quiet speech still averages well above room tone.
|
|
206
|
+
*/
|
|
207
|
+
export function analyze(
|
|
208
|
+
transcript: Transcript,
|
|
209
|
+
silences: Span[],
|
|
210
|
+
duration: number,
|
|
211
|
+
levels?: Pick<LevelStats, "windowsDb" | "windowSec" | "speechDb">,
|
|
212
|
+
): Analysis {
|
|
213
|
+
const words = transcript.words;
|
|
214
|
+
const gaps: Span[] = [];
|
|
215
|
+
if (words.length > 0) {
|
|
216
|
+
const first = words[0]!;
|
|
217
|
+
const last = words[words.length - 1]!;
|
|
218
|
+
if (first.start > 0) gaps.push({ start: 0, end: first.start });
|
|
219
|
+
for (let i = 0; i < words.length - 1; i++) {
|
|
220
|
+
const a = words[i]!;
|
|
221
|
+
const b = words[i + 1]!;
|
|
222
|
+
if (b.start > a.end) gaps.push({ start: a.end, end: b.start });
|
|
223
|
+
}
|
|
224
|
+
if (duration > last.end) gaps.push({ start: last.end, end: duration });
|
|
225
|
+
} else {
|
|
226
|
+
gaps.push({ start: 0, end: duration });
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
const bounded = silences.map((s) => ({ start: s.start, end: Math.min(s.end, duration) }));
|
|
230
|
+
|
|
231
|
+
const fillers = words.flatMap((w, wordIndex) => {
|
|
232
|
+
const norm = normalizeToken(w.text);
|
|
233
|
+
return FILLER_WORDS.has(norm) ? [{ wordIndex, text: w.text, start: w.start, end: w.end }] : [];
|
|
234
|
+
});
|
|
235
|
+
const fillerIndices = new Set(fillers.map((f) => f.wordIndex));
|
|
236
|
+
|
|
237
|
+
const cuttable: Span[] = [];
|
|
238
|
+
for (const sil of bounded) {
|
|
239
|
+
if (sil.end - sil.start < MIN_CUTTABLE) continue;
|
|
240
|
+
// Veto: a non-filler word wholly inside the silence means the two signals
|
|
241
|
+
// disagree about where speech is. Keep that word and cut around it…
|
|
242
|
+
let conflicts = words.filter(
|
|
243
|
+
(w, i) => !fillerIndices.has(i) && w.start >= sil.start && w.end <= sil.end && w.end > w.start,
|
|
244
|
+
);
|
|
245
|
+
// …unless the region is measurably dead air, in which case the transcript
|
|
246
|
+
// is the signal that's wrong.
|
|
247
|
+
if (conflicts.length > 0 && levels && levels.windowsDb.length > 0) {
|
|
248
|
+
const meanDb = regionLevelDb(levels.windowsDb, levels.windowSec, sil.start, sil.end);
|
|
249
|
+
if (meanDb <= levels.speechDb - DEAD_AIR_DROP) conflicts = [];
|
|
250
|
+
}
|
|
251
|
+
for (const piece of subtractSpans(sil, conflicts)) {
|
|
252
|
+
if (piece.end - piece.start >= MIN_CUTTABLE) cuttable.push(piece);
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
cuttable.sort((a, b) => a.start - b.start);
|
|
256
|
+
|
|
257
|
+
return { silences: bounded, gaps, cuttable, breaths: detectBreaths(levels, duration), fillers };
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
/** A dip must last at least this long to be a breath and not a plosive gap. */
|
|
261
|
+
const MIN_BREATH_SEC = 0.12;
|
|
262
|
+
/** Dips this quiet are pauses; the silencedetect threshold is stricter still. */
|
|
263
|
+
const BREATH_DROP = 10;
|
|
264
|
+
|
|
265
|
+
/**
|
|
266
|
+
* Sub-silence pauses — where a speaker draws breath between phrases.
|
|
267
|
+
*
|
|
268
|
+
* `silences` cannot serve this purpose: `silencedetect` runs with a 0.35 s
|
|
269
|
+
* minimum (see `detectSilences`), while real inter-phrase breaths are
|
|
270
|
+
* 120–300 ms. On the reference take that floor left 8 silences across 68 s,
|
|
271
|
+
* which is why anything driven off them degenerates to uniform pacing.
|
|
272
|
+
*
|
|
273
|
+
* The 100 ms RMS series is already measured for threshold derivation, so this
|
|
274
|
+
* costs no extra ffmpeg pass: a run of windows sitting `BREATH_DROP` below
|
|
275
|
+
* the take's own speech level is a pause, whatever the transcript claims.
|
|
276
|
+
* These are phrase boundaries the word stamps cannot provide — whisper `-ml 1`
|
|
277
|
+
* emits contiguous stamps, so inter-word gaps do not exist (PHASE0 "Signal
|
|
278
|
+
* fusion", FINDINGS §18).
|
|
279
|
+
*/
|
|
280
|
+
export function detectBreaths(
|
|
281
|
+
levels: Pick<LevelStats, "windowsDb" | "windowSec" | "speechDb"> | undefined,
|
|
282
|
+
duration: number,
|
|
283
|
+
): Span[] {
|
|
284
|
+
if (!levels || levels.windowsDb.length === 0) return [];
|
|
285
|
+
const threshold = levels.speechDb - BREATH_DROP;
|
|
286
|
+
const breaths: Span[] = [];
|
|
287
|
+
let runStart: number | null = null;
|
|
288
|
+
for (let i = 0; i <= levels.windowsDb.length; i++) {
|
|
289
|
+
const quiet = i < levels.windowsDb.length && levels.windowsDb[i]! <= threshold;
|
|
290
|
+
if (quiet && runStart === null) runStart = i;
|
|
291
|
+
if (!quiet && runStart !== null) {
|
|
292
|
+
const start = runStart * levels.windowSec;
|
|
293
|
+
const end = Math.min(i * levels.windowSec, duration);
|
|
294
|
+
if (end - start >= MIN_BREATH_SEC) breaths.push({ start, end });
|
|
295
|
+
runStart = null;
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
return breaths;
|
|
299
|
+
}
|
package/src/assemble.ts
ADDED
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
import type { Transcript } from "./schema";
|
|
2
|
+
import type { Scene, SceneCue } from "./scene-schema";
|
|
3
|
+
import { resolveSceneProps } from "./scene-registry";
|
|
4
|
+
import type { TimeMap } from "./timemap";
|
|
5
|
+
|
|
6
|
+
/** Scenes shorter than this on screen get extended; still shorter → dropped. */
|
|
7
|
+
const MIN_SCENE_SEC = 1.2;
|
|
8
|
+
const DROP_BELOW_SEC = 0.8;
|
|
9
|
+
/**
|
|
10
|
+
* On a short take a 2s graphic reads as a flicker, and there is no later beat
|
|
11
|
+
* to make up for it — hold every surviving scene longer (FINDINGS §29).
|
|
12
|
+
*/
|
|
13
|
+
const SHORT_TAKE_SEC = 45;
|
|
14
|
+
const SHORT_TAKE_MIN_SCENE_SEC = 3;
|
|
15
|
+
/**
|
|
16
|
+
* A graphic punches in, makes its point, and hands the frame back to the
|
|
17
|
+
* speaker — it does not have to span the moment that motivated it. Keeps the
|
|
18
|
+
* §4.5 pattern-interrupt rhythm instead of 10s static cards (FINDINGS §3).
|
|
19
|
+
* Exported: the beat scheduler budgets coverage with this same number (§7).
|
|
20
|
+
*/
|
|
21
|
+
export const MAX_SCENE_SEC = 5;
|
|
22
|
+
/**
|
|
23
|
+
* A lower third never TAKES the frame — the speaker stays full-bleed the
|
|
24
|
+
* whole time — so the pattern-interrupt argument behind MAX_SCENE_SEC does
|
|
25
|
+
* not apply and the punch-out at 5s was cutting cards off mid-sentence
|
|
26
|
+
* (R20 §95, seen on the first real landscape run). It holds through its
|
|
27
|
+
* whole moment instead, under this generous ceiling so a rambling moment
|
|
28
|
+
* still cannot pin one static card up for a minute.
|
|
29
|
+
*/
|
|
30
|
+
export const MAX_OVERLAY_SCENE_SEC = 15;
|
|
31
|
+
|
|
32
|
+
/** The on-screen cap for a cue, by how much frame its layout takes. */
|
|
33
|
+
const maxSceneSecFor = (layout: SceneCue["layout"]): number =>
|
|
34
|
+
layout === "lower-third" ? MAX_OVERLAY_SCENE_SEC : MAX_SCENE_SEC;
|
|
35
|
+
/** Breathing room enforced between consecutive scenes. */
|
|
36
|
+
const SCENE_GAP_SEC = 0.05;
|
|
37
|
+
|
|
38
|
+
export interface AssembleResult {
|
|
39
|
+
cues: SceneCue[];
|
|
40
|
+
dropped: Array<{ id: string; reason: string }>;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Resolve word-anchored scenes into output-timed cues (PHASE1 §5):
|
|
45
|
+
* anchors → output time via the TimeMap (scenes whose words were cut are
|
|
46
|
+
* dropped), props resolved as defaults←props←overrides, then sorted,
|
|
47
|
+
* de-overlapped and given a minimum on-screen duration. Gaps are implicit
|
|
48
|
+
* full-bleed — the stage defaults to the talking head when no cue is active.
|
|
49
|
+
*/
|
|
50
|
+
export function assembleScenes(
|
|
51
|
+
scenes: readonly Scene[],
|
|
52
|
+
transcript: Transcript,
|
|
53
|
+
map: TimeMap,
|
|
54
|
+
): AssembleResult {
|
|
55
|
+
const dropped: AssembleResult["dropped"] = [];
|
|
56
|
+
const resolved: SceneCue[] = [];
|
|
57
|
+
|
|
58
|
+
for (const scene of scenes) {
|
|
59
|
+
const words = transcript.words.slice(scene.anchor.startWord, scene.anchor.endWord + 1);
|
|
60
|
+
if (words.length === 0) {
|
|
61
|
+
dropped.push({ id: scene.id, reason: "anchor out of transcript range" });
|
|
62
|
+
continue;
|
|
63
|
+
}
|
|
64
|
+
const mapped = words
|
|
65
|
+
.map((w) => map.mapWord(w))
|
|
66
|
+
.filter((m): m is { start: number; end: number } => m !== null);
|
|
67
|
+
if (mapped.length === 0) {
|
|
68
|
+
dropped.push({ id: scene.id, reason: "anchor words were entirely cut" });
|
|
69
|
+
continue;
|
|
70
|
+
}
|
|
71
|
+
const props = resolveSceneProps(scene.component, scene.props, scene.overrides);
|
|
72
|
+
if (props === null) {
|
|
73
|
+
dropped.push({ id: scene.id, reason: `invalid props for ${scene.component}` });
|
|
74
|
+
continue;
|
|
75
|
+
}
|
|
76
|
+
resolved.push({
|
|
77
|
+
id: scene.id,
|
|
78
|
+
layout: scene.layout,
|
|
79
|
+
component: scene.component,
|
|
80
|
+
props,
|
|
81
|
+
startSec: Math.min(...mapped.map((m) => m.start)),
|
|
82
|
+
endSec: Math.max(...mapped.map((m) => m.end)),
|
|
83
|
+
});
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
resolved.sort((a, b) => a.startSec - b.startSec);
|
|
87
|
+
|
|
88
|
+
// Never let the floor eat the video: a 3s minimum is right on a 30s take and
|
|
89
|
+
// absurd on a 6s one, so it is also capped at a share of the runtime.
|
|
90
|
+
const minScene =
|
|
91
|
+
map.outputDuration < SHORT_TAKE_SEC
|
|
92
|
+
? Math.max(MIN_SCENE_SEC, Math.min(SHORT_TAKE_MIN_SCENE_SEC, map.outputDuration * 0.15))
|
|
93
|
+
: MIN_SCENE_SEC;
|
|
94
|
+
|
|
95
|
+
// Scenes are exclusive — one stage state at a time.
|
|
96
|
+
const cues: SceneCue[] = [];
|
|
97
|
+
for (const cue of resolved) {
|
|
98
|
+
const prev = cues[cues.length - 1];
|
|
99
|
+
if (prev && cue.startSec < prev.endSec + SCENE_GAP_SEC) {
|
|
100
|
+
cue.startSec = prev.endSec + SCENE_GAP_SEC;
|
|
101
|
+
}
|
|
102
|
+
if (cue.endSec - cue.startSec < minScene) {
|
|
103
|
+
cue.endSec = cue.startSec + minScene;
|
|
104
|
+
}
|
|
105
|
+
cue.endSec = Math.min(cue.endSec, cue.startSec + maxSceneSecFor(cue.layout), map.outputDuration);
|
|
106
|
+
if (cue.endSec - cue.startSec < DROP_BELOW_SEC) {
|
|
107
|
+
dropped.push({ id: cue.id, reason: "too short after clamping" });
|
|
108
|
+
continue;
|
|
109
|
+
}
|
|
110
|
+
cues.push(cue);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// The min-duration extension above can re-introduce overlap with the NEXT
|
|
114
|
+
// cue's start; walk once more and trim forward.
|
|
115
|
+
for (let i = 0; i < cues.length - 1; i++) {
|
|
116
|
+
const cur = cues[i]!;
|
|
117
|
+
const next = cues[i + 1]!;
|
|
118
|
+
if (next.startSec < cur.endSec + SCENE_GAP_SEC) {
|
|
119
|
+
cur.endSec = Math.max(cur.startSec + DROP_BELOW_SEC, next.startSec - SCENE_GAP_SEC);
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
return { cues, dropped };
|
|
124
|
+
}
|
package/src/browser.ts
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Browser-safe surface of @ossclip/core — everything the Remotion bundle may
|
|
3
|
+
* import AT RUNTIME. No node built-ins, no SDKs, no fs/child_process anywhere
|
|
4
|
+
* in this module graph (scene-schema + scene-registry + zod only); the rest of
|
|
5
|
+
* core is re-exported as types, which erase at compile time.
|
|
6
|
+
*/
|
|
7
|
+
export * from "./scene-schema";
|
|
8
|
+
export * from "./scene-registry";
|
|
9
|
+
export * from "./overrides";
|
|
10
|
+
// The editor derives its plain takes with the SAME function the pipeline
|
|
11
|
+
// uses — a copy would drift and the two timelines would disagree.
|
|
12
|
+
export * from "./fill";
|
|
13
|
+
export { ZOOM_MAX_SCALE, zoomScaleAt, type ZoomSegment } from "./zoom";
|
|
14
|
+
// Pure geometry only — the ffmpeg/cache half lives in ./content-rect-detect
|
|
15
|
+
// and must never enter the Remotion bundle.
|
|
16
|
+
export {
|
|
17
|
+
contentRectAt,
|
|
18
|
+
cropFilter,
|
|
19
|
+
type ContentRect,
|
|
20
|
+
type ContentRectSegment,
|
|
21
|
+
} from "./content-rect";
|
|
22
|
+
export type { CaptionLine, CaptionWord } from "./captions";
|
|
23
|
+
export type { KeptSpan } from "./timemap";
|
|
24
|
+
export type { Probe, Production, RenderSettings, Segment, Transcript, Word } from "./schema";
|
package/src/captions.ts
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
import type { Transcript, Word } from "./schema";
|
|
2
|
+
import type { TimeMap } from "./timemap";
|
|
3
|
+
|
|
4
|
+
/** Caption timing lives in OUTPUT time — captions never know about cuts. */
|
|
5
|
+
export interface CaptionWord {
|
|
6
|
+
text: string;
|
|
7
|
+
start: number;
|
|
8
|
+
end: number;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
export interface CaptionLine {
|
|
12
|
+
words: CaptionWord[];
|
|
13
|
+
start: number;
|
|
14
|
+
end: number;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
export interface CaptionOptions {
|
|
18
|
+
maxWordsPerLine?: number;
|
|
19
|
+
maxLineDuration?: number;
|
|
20
|
+
/** A speech gap longer than this starts a fresh line. */
|
|
21
|
+
maxGap?: number;
|
|
22
|
+
/** How long a line lingers after its last word (clamped to the next line). */
|
|
23
|
+
hold?: number;
|
|
24
|
+
/**
|
|
25
|
+
* Output times a line must not span — scene-cue starts/ends. The caption
|
|
26
|
+
* anchor is resolved once per line from the layout at its start, so a line
|
|
27
|
+
* crossing a layout boundary would sit in the WRONG layout's band and can
|
|
28
|
+
* land on a card or the face (FINDINGS §6b). Lines flush at boundaries and
|
|
29
|
+
* their hold never extends past one.
|
|
30
|
+
*/
|
|
31
|
+
breakpoints?: number[];
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export function buildCaptionLines(
|
|
35
|
+
transcript: Transcript,
|
|
36
|
+
map: TimeMap,
|
|
37
|
+
opts: CaptionOptions = {},
|
|
38
|
+
): CaptionLine[] {
|
|
39
|
+
const maxWords = opts.maxWordsPerLine ?? 3;
|
|
40
|
+
const maxDur = opts.maxLineDuration ?? 1.2;
|
|
41
|
+
const maxGap = opts.maxGap ?? 0.6;
|
|
42
|
+
const hold = opts.hold ?? 0.35;
|
|
43
|
+
const breakpoints = [...(opts.breakpoints ?? [])].sort((a, b) => a - b);
|
|
44
|
+
|
|
45
|
+
const mapped: CaptionWord[] = [];
|
|
46
|
+
for (const w of transcript.words) {
|
|
47
|
+
const m = map.mapWord(w as Word);
|
|
48
|
+
if (m) mapped.push({ text: w.text, start: m.start, end: m.end });
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
const lines: CaptionLine[] = [];
|
|
52
|
+
let current: CaptionWord[] = [];
|
|
53
|
+
const flush = () => {
|
|
54
|
+
if (current.length === 0) return;
|
|
55
|
+
lines.push({
|
|
56
|
+
words: current,
|
|
57
|
+
start: current[0]!.start,
|
|
58
|
+
end: current[current.length - 1]!.end,
|
|
59
|
+
});
|
|
60
|
+
current = [];
|
|
61
|
+
};
|
|
62
|
+
|
|
63
|
+
for (const w of mapped) {
|
|
64
|
+
const lineStart = current[0]?.start ?? w.start;
|
|
65
|
+
const prevEnd = current[current.length - 1]?.end;
|
|
66
|
+
const crossesBoundary =
|
|
67
|
+
current.length > 0 && breakpoints.some((b) => b > lineStart + 1e-6 && b <= w.start + 1e-6);
|
|
68
|
+
if (
|
|
69
|
+
current.length >= maxWords ||
|
|
70
|
+
w.end - lineStart > maxDur ||
|
|
71
|
+
(prevEnd !== undefined && w.start - prevEnd > maxGap) ||
|
|
72
|
+
crossesBoundary
|
|
73
|
+
) {
|
|
74
|
+
flush();
|
|
75
|
+
}
|
|
76
|
+
current.push(w);
|
|
77
|
+
}
|
|
78
|
+
flush();
|
|
79
|
+
|
|
80
|
+
for (let i = 0; i < lines.length; i++) {
|
|
81
|
+
const line = lines[i]!;
|
|
82
|
+
const next = lines[i + 1];
|
|
83
|
+
const lastWordEnd = line.words[line.words.length - 1]!.end;
|
|
84
|
+
const boundary = breakpoints.find((b) => b > line.start + 1e-6);
|
|
85
|
+
let end = line.end + hold;
|
|
86
|
+
if (next) end = Math.min(end, next.start);
|
|
87
|
+
if (boundary !== undefined) end = Math.min(end, boundary);
|
|
88
|
+
// A single word physically spanning a boundary stays readable to its end.
|
|
89
|
+
line.end = Math.min(Math.max(end, lastWordEnd), map.outputDuration);
|
|
90
|
+
}
|
|
91
|
+
return lines;
|
|
92
|
+
}
|