@ossclip/core 0.1.24 → 0.1.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/fonts/NotoNastaliqUrdu-Bold.ttf +0 -0
- package/assets/fonts/OFL.txt +93 -0
- package/assets/fonts/README.md +15 -0
- package/package.json +2 -1
- package/src/blooper.ts +91 -8
- package/src/browser.ts +8 -0
- package/src/captions.ts +45 -2
- package/src/concat.ts +5 -5
- package/src/config.ts +110 -0
- package/src/content-rect-detect.ts +16 -4
- package/src/content-rect.ts +211 -0
- package/src/cover.ts +21 -5
- package/src/cutlist.ts +38 -6
- package/src/dictionary.ts +56 -0
- package/src/export-premiere-project.ts +26 -9
- package/src/fonts.ts +17 -0
- package/src/index.ts +3 -0
- package/src/ingest.ts +88 -2
- package/src/normalize.ts +273 -127
- package/src/producer/index.ts +1 -0
- package/src/producer/repair.ts +36 -13
- package/src/producer/youtube.ts +434 -0
- package/src/retake.ts +104 -2
- package/src/scene-schema.ts +10 -0
- package/src/thumbnail.ts +412 -0
- package/src/transcribe.ts +22 -0
- package/src/zoom.ts +63 -12
|
Binary file
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
Copyright 2022 The Noto Project Authors (https://github.com/notofonts/nastaliq)
|
|
2
|
+
|
|
3
|
+
This Font Software is licensed under the SIL Open Font License, Version 1.1.
|
|
4
|
+
This license is copied below, and is also available with a FAQ at:
|
|
5
|
+
https://scripts.sil.org/OFL
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
-----------------------------------------------------------
|
|
9
|
+
SIL OPEN FONT LICENSE Version 1.1 - 26 February 2007
|
|
10
|
+
-----------------------------------------------------------
|
|
11
|
+
|
|
12
|
+
PREAMBLE
|
|
13
|
+
The goals of the Open Font License (OFL) are to stimulate worldwide
|
|
14
|
+
development of collaborative font projects, to support the font creation
|
|
15
|
+
efforts of academic and linguistic communities, and to provide a free and
|
|
16
|
+
open framework in which fonts may be shared and improved in partnership
|
|
17
|
+
with others.
|
|
18
|
+
|
|
19
|
+
The OFL allows the licensed fonts to be used, studied, modified and
|
|
20
|
+
redistributed freely as long as they are not sold by themselves. The
|
|
21
|
+
fonts, including any derivative works, can be bundled, embedded,
|
|
22
|
+
redistributed and/or sold with any software provided that any reserved
|
|
23
|
+
names are not used by derivative works. The fonts and derivatives,
|
|
24
|
+
however, cannot be released under any other type of license. The
|
|
25
|
+
requirement for fonts to remain under this license does not apply
|
|
26
|
+
to any document created using the fonts or their derivatives.
|
|
27
|
+
|
|
28
|
+
DEFINITIONS
|
|
29
|
+
"Font Software" refers to the set of files released by the Copyright
|
|
30
|
+
Holder(s) under this license and clearly marked as such. This may
|
|
31
|
+
include source files, build scripts and documentation.
|
|
32
|
+
|
|
33
|
+
"Reserved Font Name" refers to any names specified as such after the
|
|
34
|
+
copyright statement(s).
|
|
35
|
+
|
|
36
|
+
"Original Version" refers to the collection of Font Software components as
|
|
37
|
+
distributed by the Copyright Holder(s).
|
|
38
|
+
|
|
39
|
+
"Modified Version" refers to any derivative made by adding to, deleting,
|
|
40
|
+
or substituting -- in part or in whole -- any of the components of the
|
|
41
|
+
Original Version, by changing formats or by porting the Font Software to a
|
|
42
|
+
new environment.
|
|
43
|
+
|
|
44
|
+
"Author" refers to any designer, engineer, programmer, technical
|
|
45
|
+
writer or other person who contributed to the Font Software.
|
|
46
|
+
|
|
47
|
+
PERMISSION & CONDITIONS
|
|
48
|
+
Permission is hereby granted, free of charge, to any person obtaining
|
|
49
|
+
a copy of the Font Software, to use, study, copy, merge, embed, modify,
|
|
50
|
+
redistribute, and sell modified and unmodified copies of the Font
|
|
51
|
+
Software, subject to the following conditions:
|
|
52
|
+
|
|
53
|
+
1) Neither the Font Software nor any of its individual components,
|
|
54
|
+
in Original or Modified Versions, may be sold by itself.
|
|
55
|
+
|
|
56
|
+
2) Original or Modified Versions of the Font Software may be bundled,
|
|
57
|
+
redistributed and/or sold with any software, provided that each copy
|
|
58
|
+
contains the above copyright notice and this license. These can be
|
|
59
|
+
included either as stand-alone text files, human-readable headers or
|
|
60
|
+
in the appropriate machine-readable metadata fields within text or
|
|
61
|
+
binary files as long as those fields can be easily viewed by the user.
|
|
62
|
+
|
|
63
|
+
3) No Modified Version of the Font Software may use the Reserved Font
|
|
64
|
+
Name(s) unless explicit written permission is granted by the corresponding
|
|
65
|
+
Copyright Holder. This restriction only applies to the primary font name as
|
|
66
|
+
presented to the users.
|
|
67
|
+
|
|
68
|
+
4) The name(s) of the Copyright Holder(s) or the Author(s) of the Font
|
|
69
|
+
Software shall not be used to promote, endorse or advertise any
|
|
70
|
+
Modified Version, except to acknowledge the contribution(s) of the
|
|
71
|
+
Copyright Holder(s) and the Author(s) or with their explicit written
|
|
72
|
+
permission.
|
|
73
|
+
|
|
74
|
+
5) The Font Software, modified or unmodified, in part or in whole,
|
|
75
|
+
must be distributed entirely under this license, and must not be
|
|
76
|
+
distributed under any other license. The requirement for fonts to
|
|
77
|
+
remain under this license does not apply to any document created
|
|
78
|
+
using the Font Software.
|
|
79
|
+
|
|
80
|
+
TERMINATION
|
|
81
|
+
This license becomes null and void if any of the above conditions are
|
|
82
|
+
not met.
|
|
83
|
+
|
|
84
|
+
DISCLAIMER
|
|
85
|
+
THE FONT SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
|
86
|
+
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO ANY WARRANTIES OF
|
|
87
|
+
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT
|
|
88
|
+
OF COPYRIGHT, PATENT, TRADEMARK, OR OTHER RIGHT. IN NO EVENT SHALL THE
|
|
89
|
+
COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
|
90
|
+
INCLUDING ANY GENERAL, SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL
|
|
91
|
+
DAMAGES, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
|
92
|
+
FROM, OUT OF THE USE OR INABILITY TO USE THE FONT SOFTWARE OR FROM
|
|
93
|
+
OTHER DEALINGS IN THE FONT SOFTWARE.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# Bundled fonts
|
|
2
|
+
|
|
3
|
+
## NotoNastaliqUrdu-Bold.ttf
|
|
4
|
+
|
|
5
|
+
Noto Nastaliq Urdu Bold, from the notofonts/nastaliq **v3.007** GitHub
|
|
6
|
+
release (`googlefonts/ttf/NotoNastaliqUrdu-Bold.ttf`), © 2022 The Noto
|
|
7
|
+
Project Authors — licensed under the SIL Open Font License 1.1 (`OFL.txt`
|
|
8
|
+
beside this file). The OFL permits bundling and redistribution with the
|
|
9
|
+
license text included, which is why the license ships in this directory.
|
|
10
|
+
|
|
11
|
+
Why it is bundled at all: caption rendering must be deterministic across
|
|
12
|
+
machines, and a Linux CI box (or a fresh Windows install) has no Nastaliq
|
|
13
|
+
face — Urdu captions there fell back to whatever Arabic-script font the OS
|
|
14
|
+
had, or to tofu. Only the Bold face ships because captions render at
|
|
15
|
+
`fontWeight: 900` and one face keeps the package small (~616 KB).
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ossclip/core",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.25",
|
|
4
4
|
"description": "ossclip's framework-free pipeline: schema, transcription, analysis, cutlist, captions, framing, and the LLM producer",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -20,6 +20,7 @@
|
|
|
20
20
|
],
|
|
21
21
|
"dependencies": {
|
|
22
22
|
"@anthropic-ai/sdk": "^0.115.0",
|
|
23
|
+
"@google/genai": "^2.17.1",
|
|
23
24
|
"zod": "^3.25.0"
|
|
24
25
|
},
|
|
25
26
|
"homepage": "https://github.com/AhsanAyaz/ossclip#readme",
|
package/src/blooper.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { normalizeToken } from "./analyze";
|
|
2
2
|
import { isSentenceStart } from "./clip";
|
|
3
3
|
import { levenshtein } from "./phonetics";
|
|
4
|
-
import type { Transcript } from "./schema";
|
|
4
|
+
import type { Span, Transcript } from "./schema";
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* Blooper removal by SPOKEN MARKER (R27 §122).
|
|
@@ -55,8 +55,40 @@ export interface BloopSpan {
|
|
|
55
55
|
* silently cutting a good take (Task 3, editor-dogfood-fixes plan).
|
|
56
56
|
*/
|
|
57
57
|
matched: string[];
|
|
58
|
+
/**
|
|
59
|
+
* True when the sentence-start backscan hit MAX_WALKBACK_SEC before it
|
|
60
|
+
* found punctuation — the span was cut short of a real sentence boundary
|
|
61
|
+
* and the report must say so out loud (`formatBloopSpan`).
|
|
62
|
+
*/
|
|
63
|
+
truncated?: boolean;
|
|
58
64
|
}
|
|
59
65
|
|
|
66
|
+
/**
|
|
67
|
+
* How far past the marker word's STAMPED end a silence may start and still be
|
|
68
|
+
* treated as the marker's own trailing dead air. Whisper's `-ml 1` stamps
|
|
69
|
+
* stretch over pauses (§18, PHASE1-FINDINGS.md), so the stamped end of a
|
|
70
|
+
* spoken marker routinely lands BEFORE its acoustic end — the 2026-08-16
|
|
71
|
+
* incident: "blooper." stamped to end at 670.0 while the audio ran to ~670.4
|
|
72
|
+
* (proven by the bracketing silences 668.09–669.3 and 670.4–671.68), so 0.4s
|
|
73
|
+
* of audible "blooper" leaked into the output at ~10:02. Extending the cut
|
|
74
|
+
* through any silence starting within this window swallows the acoustic tail
|
|
75
|
+
* — and the debris sliver ("And", 670.0–670.4) whisper stamped between the
|
|
76
|
+
* marker and the pause. 0.75s is deliberately wider than the observed 0.4s
|
|
77
|
+
* stretch but well under the shortest gap a speaker leaves before a real
|
|
78
|
+
* next take.
|
|
79
|
+
*/
|
|
80
|
+
const MAX_MARKER_BLEED_SEC = 0.75;
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Ceiling on the sentence-start backscan, in source seconds. An ASR stretch
|
|
84
|
+
* with no terminal punctuation (rambling delivery, a non-English fine-tune,
|
|
85
|
+
* a hallucinated run-on) lets the scan walk back through MINUTES of good
|
|
86
|
+
* take from one spoken marker — the same failure shape as §133's 7.08s fuzzy
|
|
87
|
+
* cut, unbounded. 30s is longer than any real single-sentence flub and short
|
|
88
|
+
* enough that a capped cut is reviewable in the report.
|
|
89
|
+
*/
|
|
90
|
+
const MAX_WALKBACK_SEC = 30;
|
|
91
|
+
|
|
60
92
|
// Fuzzy matching only turns on once the marker is long enough that a false
|
|
61
93
|
// positive is unlikely — a short marker like "cut" sound-alikes ("cat") and
|
|
62
94
|
// sits within edit distance 2 of half the dictionary ("but", "gut", "cot"),
|
|
@@ -120,9 +152,18 @@ function isPluralPair(a: string, b: string): boolean {
|
|
|
120
152
|
* marker of at least `FUZZY_MIN_MARKER_LEN` characters also matches an ASR
|
|
121
153
|
* mishearing — see `matchMarker`.
|
|
122
154
|
*
|
|
155
|
+
* `silences` (source seconds, `analysis.silences`' shape) lets a span's end
|
|
156
|
+
* extend through the marker's own trailing dead air — the §18 stamp-stretch
|
|
157
|
+
* bleed `MAX_MARKER_BLEED_SEC` documents. Omitting it reproduces the
|
|
158
|
+
* stamped-end behavior exactly.
|
|
159
|
+
*
|
|
123
160
|
* Returns spans in transcript order, non-overlapping.
|
|
124
161
|
*/
|
|
125
|
-
export function findBloopSpans(
|
|
162
|
+
export function findBloopSpans(
|
|
163
|
+
transcript: Transcript,
|
|
164
|
+
marker: string,
|
|
165
|
+
silences?: readonly Span[],
|
|
166
|
+
): BloopSpan[] {
|
|
126
167
|
const want = normalizeToken(marker);
|
|
127
168
|
if (!want) return [];
|
|
128
169
|
const words = transcript.words;
|
|
@@ -138,31 +179,67 @@ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpa
|
|
|
138
179
|
|
|
139
180
|
// Walk back over the attempt this marker spoiled, to the start of its
|
|
140
181
|
// sentence. The marker's own text usually ENDS a sentence ("blooper."), so
|
|
141
|
-
// the scan starts at the word before it.
|
|
182
|
+
// the scan starts at the word before it. Capped at MAX_WALKBACK_SEC of
|
|
183
|
+
// source time (rationale on the constant); a capped span is flagged so
|
|
184
|
+
// the report can shout about it.
|
|
142
185
|
let start = i;
|
|
143
|
-
|
|
186
|
+
let truncated = false;
|
|
187
|
+
while (start > 0 && !isSentenceStart(transcript, start)) {
|
|
188
|
+
if (words[i]!.end - words[start - 1]!.start > MAX_WALKBACK_SEC) {
|
|
189
|
+
truncated = true;
|
|
190
|
+
break;
|
|
191
|
+
}
|
|
192
|
+
start--;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
// Extend the end through the marker's trailing dead air (2026-08-16
|
|
196
|
+
// incident, see MAX_MARKER_BLEED_SEC): repeatedly absorb any silence that
|
|
197
|
+
// starts within the bleed window and ends past the current end. The loop
|
|
198
|
+
// re-scans because absorbing one silence can bring the next within reach
|
|
199
|
+
// — the chained-silence shape.
|
|
200
|
+
let endSec = words[i]!.end;
|
|
201
|
+
if (silences && silences.length > 0) {
|
|
202
|
+
let extended = true;
|
|
203
|
+
while (extended) {
|
|
204
|
+
extended = false;
|
|
205
|
+
for (const s of silences) {
|
|
206
|
+
if (s.start <= endSec + MAX_MARKER_BLEED_SEC && s.end > endSec) {
|
|
207
|
+
endSec = s.end;
|
|
208
|
+
extended = true;
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
}
|
|
144
213
|
|
|
145
214
|
// Consecutive attempts: if everything between the previous span's end and
|
|
146
215
|
// this attempt's start is already being dropped, merge rather than leave a
|
|
147
|
-
// one-word island of a sentence nobody finished.
|
|
216
|
+
// one-word island of a sentence nobody finished. The seconds arm exists
|
|
217
|
+
// because the previous span's EXTENDED end can reach past word stamps the
|
|
218
|
+
// index arm never sees — the incident's "And" debris sat between two
|
|
219
|
+
// attempts that were not word-index adjacent once the first end grew.
|
|
148
220
|
const prev = spans[spans.length - 1];
|
|
149
221
|
let markers = 1;
|
|
150
222
|
let matched = match.exact ? [] : [match.surface];
|
|
151
|
-
if (prev && start <= prev.endWord + 1) {
|
|
223
|
+
if (prev && (start <= prev.endWord + 1 || words[start]!.start <= prev.endSec)) {
|
|
152
224
|
spans.pop();
|
|
153
225
|
start = prev.startWord;
|
|
154
226
|
markers = prev.markers + 1;
|
|
155
227
|
matched = [...prev.matched, ...matched];
|
|
228
|
+
truncated = truncated || prev.truncated === true;
|
|
229
|
+
// A merged span must never shrink: the previous extension already
|
|
230
|
+
// proved that audio dead.
|
|
231
|
+
endSec = Math.max(endSec, prev.endSec);
|
|
156
232
|
}
|
|
157
233
|
|
|
158
234
|
spans.push({
|
|
159
235
|
startWord: start,
|
|
160
236
|
endWord: i,
|
|
161
237
|
startSec: words[start]!.start,
|
|
162
|
-
endSec
|
|
238
|
+
endSec,
|
|
163
239
|
markers,
|
|
164
240
|
marker: want,
|
|
165
241
|
matched,
|
|
242
|
+
...(truncated ? { truncated: true } : {}),
|
|
166
243
|
});
|
|
167
244
|
}
|
|
168
245
|
return spans;
|
|
@@ -187,5 +264,11 @@ export function formatBloopSpan(transcript: Transcript, span: BloopSpan): string
|
|
|
187
264
|
span.matched.length > 0
|
|
188
265
|
? " " + span.matched.map((m) => `matched "${m}" ~ "${span.marker}"`).join(", ")
|
|
189
266
|
: "";
|
|
190
|
-
|
|
267
|
+
// A capped walk-back is a cut whose start the code chose by fiat, not by
|
|
268
|
+
// punctuation — the one case where the span boundary is a guess, so the
|
|
269
|
+
// report line must be loud about it.
|
|
270
|
+
const capped = span.truncated
|
|
271
|
+
? " (walk-back capped at 30s — unpunctuated stretch; check this cut)"
|
|
272
|
+
: "";
|
|
273
|
+
return `"${said}"${attempts}${fuzzy}${capped}`;
|
|
191
274
|
}
|
package/src/browser.ts
CHANGED
|
@@ -18,6 +18,7 @@ export {
|
|
|
18
18
|
cropFilter,
|
|
19
19
|
type ContentRect,
|
|
20
20
|
type ContentRectSegment,
|
|
21
|
+
type FramingSegment,
|
|
21
22
|
} from "./content-rect";
|
|
22
23
|
// lineDirection is a VALUE export but stays browser-safe: captions.ts
|
|
23
24
|
// imports types only. CaptionTrack needs it at render time (Urdu field test
|
|
@@ -27,9 +28,16 @@ export {
|
|
|
27
28
|
// a caption key from it (§137), and the editor imports this surface. Pure, and
|
|
28
29
|
// captions.ts stays browser-safe — it imports types plus timemap, which is
|
|
29
30
|
// itself type-only against ./schema.
|
|
31
|
+
// The Nastaliq trio rides the same browser-safe surface: CaptionTrack needs
|
|
32
|
+
// the family name + served path for its @font-face, and `captionsNeedNastaliq`
|
|
33
|
+
// is the ONE predicate produce and the render share for "does this caption
|
|
34
|
+
// set need the bundled font" (2026-08-17 — two conditions would drift).
|
|
30
35
|
export {
|
|
31
36
|
backfillSrcStart,
|
|
37
|
+
captionsNeedNastaliq,
|
|
32
38
|
lineDirection,
|
|
39
|
+
NASTALIQ_FONT_NAME,
|
|
40
|
+
NASTALIQ_FONT_REL,
|
|
33
41
|
type CaptionLine,
|
|
34
42
|
type CaptionWord,
|
|
35
43
|
} from "./captions";
|
package/src/captions.ts
CHANGED
|
@@ -92,6 +92,31 @@ export function lineDirection(text: string): "rtl" | "ltr" {
|
|
|
92
92
|
return "ltr";
|
|
93
93
|
}
|
|
94
94
|
|
|
95
|
+
/**
|
|
96
|
+
* The bundled Nastaliq face (Urdu captions, 2026-08-17): rendering must not
|
|
97
|
+
* depend on the render machine having an Arabic-script font — a Linux CI box
|
|
98
|
+
* has none, and macOS/Windows each substitute a DIFFERENT one, so the same
|
|
99
|
+
* render-props drew three different caption sets. The family name is what
|
|
100
|
+
* the render-side @font-face registers; the REL path is the served URL under
|
|
101
|
+
* the render's public dir, POSIX-literal like `sideImageDestRel` (produce.ts:
|
|
102
|
+
* `staticFile()` splits only on `/`, so `path.join` would break every
|
|
103
|
+
* Windows render of it).
|
|
104
|
+
*/
|
|
105
|
+
export const NASTALIQ_FONT_NAME = "Noto Nastaliq Urdu";
|
|
106
|
+
export const NASTALIQ_FONT_REL = "fonts/NotoNastaliqUrdu-Bold.ttf";
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Whether this caption set needs the bundled Nastaliq face at all — some
|
|
110
|
+
* line lays out RTL. ONE predicate shared by produce (copies the font into
|
|
111
|
+
* the public dir) and CaptionTrack (injects the @font-face): if the two
|
|
112
|
+
* sides tested different conditions, one could fetch a file the other never
|
|
113
|
+
* wrote. Pure-Latin runs — the overwhelmingly common case — copy nothing and
|
|
114
|
+
* fetch nothing, so their renders stay byte-identical.
|
|
115
|
+
*/
|
|
116
|
+
export function captionsNeedNastaliq(lines: readonly CaptionLine[]): boolean {
|
|
117
|
+
return lines.some((l) => lineDirection(l.words.map((w) => w.text).join(" ")) === "rtl");
|
|
118
|
+
}
|
|
119
|
+
|
|
95
120
|
export interface CaptionOptions {
|
|
96
121
|
maxWordsPerLine?: number;
|
|
97
122
|
maxLineDuration?: number;
|
|
@@ -109,6 +134,18 @@ export interface CaptionOptions {
|
|
|
109
134
|
breakpoints?: number[];
|
|
110
135
|
}
|
|
111
136
|
|
|
137
|
+
/**
|
|
138
|
+
* No spoken word takes this long; a stamped interval longer than it is the
|
|
139
|
+
* §18 contiguous-stamp stretch (`parseWhisperJson` clamps
|
|
140
|
+
* `next.start = w.end`, so non-speech audio is smeared into the next word's
|
|
141
|
+
* START — field case 2026-08-17: 21s of played-back video audio put "Okay,"
|
|
142
|
+
* on screen 21 seconds early). The END stamp is the trustworthy edge — clamp
|
|
143
|
+
* the start toward it. Display-only: cuts, analysis and the transcript
|
|
144
|
+
* itself never see this, and `srcStart` keeps the RAW source stamp so the
|
|
145
|
+
* §137 edit anchor does not move.
|
|
146
|
+
*/
|
|
147
|
+
export const MAX_CAPTION_WORD_LEAD_SEC = 2;
|
|
148
|
+
|
|
112
149
|
export function buildCaptionLines(
|
|
113
150
|
transcript: Transcript,
|
|
114
151
|
map: TimeMap,
|
|
@@ -124,8 +161,14 @@ export function buildCaptionLines(
|
|
|
124
161
|
for (const w of transcript.words) {
|
|
125
162
|
const m = map.mapWord(w as Word);
|
|
126
163
|
// `w.start` is source time, `m.start` output — both are needed, and only
|
|
127
|
-
// the source one is stable across a re-cut (§137).
|
|
128
|
-
|
|
164
|
+
// the source one is stable across a re-cut (§137). The display start is
|
|
165
|
+
// clamped toward the end stamp (MAX_CAPTION_WORD_LEAD_SEC above); the
|
|
166
|
+
// clamped gap to the previous word then exceeds `maxGap`, so the packer
|
|
167
|
+
// naturally breaks the line and nothing renders during the dead span.
|
|
168
|
+
if (m) {
|
|
169
|
+
const start = Math.max(m.start, m.end - MAX_CAPTION_WORD_LEAD_SEC);
|
|
170
|
+
mapped.push({ text: w.text, start, end: m.end, srcStart: w.start });
|
|
171
|
+
}
|
|
129
172
|
}
|
|
130
173
|
|
|
131
174
|
const lines: CaptionLine[] = [];
|
package/src/concat.ts
CHANGED
|
@@ -424,11 +424,11 @@ export async function concatFolder(
|
|
|
424
424
|
|
|
425
425
|
const filter = buildConcatFilter(order.length, target);
|
|
426
426
|
const inputArgs = order.flatMap((name) => ["-i", join(folder, name)]);
|
|
427
|
-
// Encode to a sibling temp path, rename only on success
|
|
428
|
-
//
|
|
429
|
-
//
|
|
430
|
-
//
|
|
431
|
-
//
|
|
427
|
+
// Encode to a sibling temp path, rename only on success (R27 §125, learned
|
|
428
|
+
// on the since-removed normalization bake): ffmpeg writes the container
|
|
429
|
+
// header as it goes, so an encode that dies mid-graph leaves a file with no
|
|
430
|
+
// `moov` atom, and a cache keyed on EXISTENCE would reuse that corpse
|
|
431
|
+
// forever. Rename is atomic on a POSIX filesystem.
|
|
432
432
|
const partial = `${outPath}.partial.mp4`;
|
|
433
433
|
try {
|
|
434
434
|
await run(tools.ffmpegPath, [
|
package/src/config.ts
CHANGED
|
@@ -3,6 +3,7 @@ import { homedir } from "node:os";
|
|
|
3
3
|
import { join } from "node:path";
|
|
4
4
|
|
|
5
5
|
import type { ModelPrice } from "./producer/usage";
|
|
6
|
+
import type { Theme } from "./scene-schema";
|
|
6
7
|
|
|
7
8
|
/** Whether a finished `produce` offers to open the editor. */
|
|
8
9
|
export type OpenEditorPref = "ask" | "always" | "never";
|
|
@@ -18,6 +19,24 @@ export interface OssclipConfig {
|
|
|
18
19
|
* sheet always uses the main model. "same" disables tiering (FINDINGS §37).
|
|
19
20
|
*/
|
|
20
21
|
fastModel?: string;
|
|
22
|
+
/**
|
|
23
|
+
* Download URLs for models the ggerganov mirror doesn't host, keyed by the
|
|
24
|
+
* bare model name — a user's own fine-tune needs one line:
|
|
25
|
+
* `"modelSources": {"my-model": "https://…/ggml-my-model.bin"}`. Wins over
|
|
26
|
+
* the curated table and the default mirror (`modelUrl` in the CLI's setup
|
|
27
|
+
* manifest). File-only like `dictionary`; validated at the consumer
|
|
28
|
+
* (`validModelSources`), so a hand-edited non-record is one warning and an
|
|
29
|
+
* ignored key, never a crash or a coerced URL.
|
|
30
|
+
*/
|
|
31
|
+
modelSources?: Record<string, string>;
|
|
32
|
+
/**
|
|
33
|
+
* Default whisper language code ("ur", "auto", …) — the durable spelling
|
|
34
|
+
* of `--whisper-language` for someone whose recordings are always in one
|
|
35
|
+
* language. The flag beats this per run, and this beats the model table's
|
|
36
|
+
* implied language. File-only like `audience`; validated at the consumer
|
|
37
|
+
* (`resolveWhisperLanguage` in produce.ts), never coerced.
|
|
38
|
+
*/
|
|
39
|
+
language?: string;
|
|
21
40
|
/**
|
|
22
41
|
* Who is in the video — "Ahsan, host of the Code with Ahsan channel".
|
|
23
42
|
* Lets the repair pass recognise a mangled proper noun instead of inventing
|
|
@@ -38,7 +57,72 @@ export interface OssclipConfig {
|
|
|
38
57
|
* `--watermark` / `--no-watermark` win over this per run.
|
|
39
58
|
*/
|
|
40
59
|
watermark?: boolean;
|
|
60
|
+
/**
|
|
61
|
+
* Terms of art the speaker uses — "JSON", "ossclip", "Genkit" — biasing
|
|
62
|
+
* transcription (whisper `--prompt`), vouching repair corrections, and
|
|
63
|
+
* canonicalizing caption casing on every run (F4, 2026-08-16: "Jason" for
|
|
64
|
+
* JSON). `--dictionary` on a run wholesale replaces this, never merges.
|
|
65
|
+
* File-only like `watermark`; validated at the consumer
|
|
66
|
+
* (`validDictionary` in produce.ts), so a hand-edited non-array is one
|
|
67
|
+
* warning and an ignored key, never a crash or a coerced term.
|
|
68
|
+
*/
|
|
69
|
+
dictionary?: string[];
|
|
70
|
+
/**
|
|
71
|
+
* Global base-theme overrides — caption/graphic colors and fonts applied to
|
|
72
|
+
* every run (F6, 2026-08-16). Partial on purpose: set `accent` alone and
|
|
73
|
+
* every other token keeps its default. Precedence per run:
|
|
74
|
+
* overrides.json (the editor's per-project doc) > this > defaultTheme.
|
|
75
|
+
* File-only; validated at the consumer (`configuredBaseTheme` in
|
|
76
|
+
* produce.ts) all-or-nothing, so one malformed key voids the whole theme
|
|
77
|
+
* with a warning instead of silently half-applying.
|
|
78
|
+
*/
|
|
79
|
+
theme?: Partial<Theme>;
|
|
80
|
+
/**
|
|
81
|
+
* Run the `--youtube` pack (SEO metadata + AI thumbnail) on every produce,
|
|
82
|
+
* so the preference is a one-time config write like `watermark`.
|
|
83
|
+
* `--youtube` / `--no-youtube` win over this per run (`resolveYoutube`).
|
|
84
|
+
*/
|
|
85
|
+
youtube?: boolean;
|
|
86
|
+
/**
|
|
87
|
+
* Path to the creator's portrait photo, fed to the AI thumbnail as the
|
|
88
|
+
* likeness reference — a path in the config like `browserExecutable`, set
|
|
89
|
+
* once rather than typed per run. `--portrait` wins over this. Existence
|
|
90
|
+
* is checked where the thumbnail is generated, not at load: an absent file
|
|
91
|
+
* there means a loud skip and the frame-grab cover stands.
|
|
92
|
+
*/
|
|
93
|
+
portrait?: string;
|
|
94
|
+
/**
|
|
95
|
+
* Who watches the channel — "junior web devs learning AI tooling". Feeds
|
|
96
|
+
* BOTH the `--youtube` pack prompt (titles/tags for the right viewer) and
|
|
97
|
+
* the AI thumbnail's concept call, so it is set once here rather than
|
|
98
|
+
* retyped per run. `--audience` wins over this per run. File-only like
|
|
99
|
+
* `portrait`; validated at the consumer (`typeof === "string"` at use),
|
|
100
|
+
* never coerced.
|
|
101
|
+
*/
|
|
102
|
+
audience?: string;
|
|
103
|
+
/**
|
|
104
|
+
* The durable thumbnail steer — "always show the terminal, never stock
|
|
105
|
+
* imagery". Fed to the AI thumbnail's concept call as a must-honor creator
|
|
106
|
+
* brief on every run. `--thumbnail-brief` wins over this per run.
|
|
107
|
+
* File-only like `audience`, validated the same way at use.
|
|
108
|
+
*/
|
|
109
|
+
thumbnailBrief?: string;
|
|
110
|
+
/**
|
|
111
|
+
* Image model for the AI thumbnail (Y3). Overrides the built-in default
|
|
112
|
+
* slug; the GEMINI_API_KEY itself stays in the environment — secrets never
|
|
113
|
+
* live in config.json (env.ts's documented rule).
|
|
114
|
+
*/
|
|
115
|
+
thumbnailModel?: string;
|
|
41
116
|
browserExecutable?: string;
|
|
117
|
+
/**
|
|
118
|
+
* Browser tabs the render runs in parallel (2026-08-17 render-speed pass).
|
|
119
|
+
* Unset means cpus-2 with a floor of 2 — the render is decode-bound, and
|
|
120
|
+
* two cores stay free for OffthreadVideo's ffmpeg extract workers. File-only
|
|
121
|
+
* like `dictionary`; validated at the consumer (`resolveRenderConcurrency`
|
|
122
|
+
* in produce.ts) as a positive integer, so a hand-edited `"4"` or `-1` is
|
|
123
|
+
* one warning and the default, never a coerced tab count.
|
|
124
|
+
*/
|
|
125
|
+
renderConcurrency?: number;
|
|
42
126
|
/**
|
|
43
127
|
* USD per million tokens, keyed by model id or family substring — overrides
|
|
44
128
|
* the built-in assumptions in `producer/usage.ts` so a run's cost line
|
|
@@ -122,6 +206,32 @@ export function loadConfig(): OssclipConfig {
|
|
|
122
206
|
// resolveWatermark), so a hand-edited non-boolean stays OFF, the safe
|
|
123
207
|
// default for a credit.
|
|
124
208
|
watermark: fileCfg.watermark,
|
|
209
|
+
// File-only for the same reason as `watermark`: these are structured
|
|
210
|
+
// values a hand-editable JSON file supplies, and parse-don't-coerce says
|
|
211
|
+
// the strict checks live at the consumer — `validDictionary` /
|
|
212
|
+
// `configuredBaseTheme` in produce.ts — where a malformed value earns a
|
|
213
|
+
// warning naming the problem and the safe default, never a coercion.
|
|
214
|
+
dictionary: fileCfg.dictionary,
|
|
215
|
+
theme: fileCfg.theme,
|
|
216
|
+
// File-only, the same posture: both are validated where they are USED —
|
|
217
|
+
// `validModelSources` / `resolveWhisperLanguage` — so a hand-edited
|
|
218
|
+
// non-record or non-string earns one warning there, never a coercion.
|
|
219
|
+
modelSources: fileCfg.modelSources,
|
|
220
|
+
language: fileCfg.language,
|
|
221
|
+
// File-only, the `watermark` posture again: `youtube` gets the strict
|
|
222
|
+
// `=== true` check at its consumer (produce's resolveYoutube), and
|
|
223
|
+
// `portrait`/`thumbnailModel` are validated where they are USED — a
|
|
224
|
+
// malformed value earns a loud skip there, never a coercion here.
|
|
225
|
+
youtube: fileCfg.youtube,
|
|
226
|
+
// File-only, the `dictionary` posture: a structured value from hand-edited
|
|
227
|
+
// JSON, validated where it is USED (`resolveRenderConcurrency`) — a
|
|
228
|
+
// malformed count earns one warning there and the cpus-2 default, never a
|
|
229
|
+
// coerced concurrency.
|
|
230
|
+
renderConcurrency: fileCfg.renderConcurrency,
|
|
231
|
+
portrait: fileCfg.portrait,
|
|
232
|
+
audience: fileCfg.audience,
|
|
233
|
+
thumbnailBrief: fileCfg.thumbnailBrief,
|
|
234
|
+
thumbnailModel: fileCfg.thumbnailModel,
|
|
125
235
|
pricing: fileCfg.pricing,
|
|
126
236
|
};
|
|
127
237
|
}
|
|
@@ -4,6 +4,7 @@ import { join } from "node:path";
|
|
|
4
4
|
import { run } from "./exec";
|
|
5
5
|
import {
|
|
6
6
|
contentRectTimeline,
|
|
7
|
+
materializeTimeline,
|
|
7
8
|
parseCropdetect,
|
|
8
9
|
pickTransition,
|
|
9
10
|
type ContentRect,
|
|
@@ -70,7 +71,7 @@ export async function detectContentRect(
|
|
|
70
71
|
timeline?: ContentRectSegment[];
|
|
71
72
|
};
|
|
72
73
|
if (cached.version === CACHE_VERSION && cached.timeline?.length) {
|
|
73
|
-
return withUniform(cached.timeline);
|
|
74
|
+
return withUniform(cached.timeline, probe);
|
|
74
75
|
}
|
|
75
76
|
}
|
|
76
77
|
|
|
@@ -95,7 +96,7 @@ export async function detectContentRect(
|
|
|
95
96
|
if (cachePath) {
|
|
96
97
|
await writeFile(cachePath, JSON.stringify({ version: CACHE_VERSION, timeline }, null, 2));
|
|
97
98
|
}
|
|
98
|
-
return withUniform(timeline);
|
|
99
|
+
return withUniform(timeline, probe);
|
|
99
100
|
}
|
|
100
101
|
|
|
101
102
|
/** How far around a coarse boundary the refinement pass looks. Must exceed the
|
|
@@ -152,8 +153,19 @@ async function refineBoundaries(
|
|
|
152
153
|
}
|
|
153
154
|
}
|
|
154
155
|
|
|
155
|
-
function withUniform(
|
|
156
|
-
|
|
156
|
+
function withUniform(
|
|
157
|
+
timeline: ContentRectSegment[],
|
|
158
|
+
probe: { width: number; height: number; duration: number },
|
|
159
|
+
): ContentRectDetection {
|
|
160
|
+
// Materialize BEFORE the single-segment uniform test, and on the cache-hit
|
|
161
|
+
// path too: the cache stores the RAW measurement, so re-judging it here is
|
|
162
|
+
// what lets the 2026-08-16 incident's cached content-rect.json re-classify
|
|
163
|
+
// as uniform on replay without a CACHE_VERSION bump (no re-measure).
|
|
164
|
+
const materialized = materializeTimeline(timeline, probe.width, probe.height, probe.duration);
|
|
165
|
+
return {
|
|
166
|
+
timeline: materialized,
|
|
167
|
+
uniform: materialized.length === 1 ? materialized[0]!.rect : null,
|
|
168
|
+
};
|
|
157
169
|
}
|
|
158
170
|
|
|
159
171
|
/** Total source seconds the framing is NOT the full frame — for reporting. */
|