@storyteller-platform/align 0.1.47 → 0.1.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/align/align.cjs +6 -1
- package/dist/align/align.js +10 -2
- package/dist/align/interpolateSentenceRanges.cjs +79 -33
- package/dist/align/interpolateSentenceRanges.d.cts +15 -2
- package/dist/align/interpolateSentenceRanges.d.ts +15 -2
- package/dist/align/interpolateSentenceRanges.js +77 -32
- package/dist/common/ffmpeg.cjs +72 -19
- package/dist/common/ffmpeg.d.cts +17 -1
- package/dist/common/ffmpeg.d.ts +17 -1
- package/dist/common/ffmpeg.js +69 -19
- package/dist/process/processAudiobook.cjs +36 -2
- package/dist/process/processAudiobook.js +42 -3
- package/package.json +2 -2
package/dist/align/align.cjs
CHANGED
|
@@ -689,17 +689,22 @@ class Aligner {
|
|
|
689
689
|
});
|
|
690
690
|
const sentenceRanges = [];
|
|
691
691
|
const chapterSentenceCounts = {};
|
|
692
|
+
const sentenceLengths = {};
|
|
692
693
|
for (const alignedChapter of audioOrderedChapters) {
|
|
693
694
|
sentenceRanges.push(...alignedChapter.sentenceRanges);
|
|
694
695
|
const sentences = await this.getChapterSentences(
|
|
695
696
|
alignedChapter.chapter.id
|
|
696
697
|
);
|
|
697
698
|
chapterSentenceCounts[alignedChapter.chapter.id] = sentences.length;
|
|
699
|
+
for (const [id, sentence] of (0, import_itertools.enumerate)(sentences)) {
|
|
700
|
+
sentenceLengths[(0, import_interpolateSentenceRanges.slotKey)({ chapterId: alignedChapter.chapter.id, id })] = sentence.text.length;
|
|
701
|
+
}
|
|
698
702
|
}
|
|
699
703
|
const interpolated = (0, import_interpolateSentenceRanges.interpolateSentenceRanges)(
|
|
700
704
|
sentenceRanges,
|
|
701
705
|
chapterSentenceCounts,
|
|
702
|
-
this.audioFileDurations
|
|
706
|
+
this.audioFileDurations,
|
|
707
|
+
sentenceLengths
|
|
703
708
|
);
|
|
704
709
|
const expanded = (0, import_getSentenceRanges.expandEmptySentenceRanges)(interpolated);
|
|
705
710
|
const collapsed = await (0, import_getSentenceRanges.collapseSentenceRangeGaps)(expanded);
|
package/dist/align/align.js
CHANGED
|
@@ -43,7 +43,10 @@ import {
|
|
|
43
43
|
getSentenceRanges,
|
|
44
44
|
mapTranscriptionTimeline
|
|
45
45
|
} from "./getSentenceRanges.js";
|
|
46
|
-
import {
|
|
46
|
+
import {
|
|
47
|
+
interpolateSentenceRanges,
|
|
48
|
+
slotKey
|
|
49
|
+
} from "./interpolateSentenceRanges.js";
|
|
47
50
|
import { findBoundaries } from "./search.js";
|
|
48
51
|
import { slugify } from "./slugify.js";
|
|
49
52
|
import { TextFragmentFactory } from "./textFragments.js";
|
|
@@ -635,17 +638,22 @@ class Aligner {
|
|
|
635
638
|
});
|
|
636
639
|
const sentenceRanges = [];
|
|
637
640
|
const chapterSentenceCounts = {};
|
|
641
|
+
const sentenceLengths = {};
|
|
638
642
|
for (const alignedChapter of audioOrderedChapters) {
|
|
639
643
|
sentenceRanges.push(...alignedChapter.sentenceRanges);
|
|
640
644
|
const sentences = await this.getChapterSentences(
|
|
641
645
|
alignedChapter.chapter.id
|
|
642
646
|
);
|
|
643
647
|
chapterSentenceCounts[alignedChapter.chapter.id] = sentences.length;
|
|
648
|
+
for (const [id, sentence] of enumerate(sentences)) {
|
|
649
|
+
sentenceLengths[slotKey({ chapterId: alignedChapter.chapter.id, id })] = sentence.text.length;
|
|
650
|
+
}
|
|
644
651
|
}
|
|
645
652
|
const interpolated = interpolateSentenceRanges(
|
|
646
653
|
sentenceRanges,
|
|
647
654
|
chapterSentenceCounts,
|
|
648
|
-
this.audioFileDurations
|
|
655
|
+
this.audioFileDurations,
|
|
656
|
+
sentenceLengths
|
|
649
657
|
);
|
|
650
658
|
const expanded = expandEmptySentenceRanges(interpolated);
|
|
651
659
|
const collapsed = await collapseSentenceRangeGaps(expanded);
|
|
@@ -18,55 +18,76 @@ var __copyProps = (to, from, except, desc) => {
|
|
|
18
18
|
var __toCommonJS = (mod) => __copyProps(__defProp({}, "__esModule", { value: true }), mod);
|
|
19
19
|
var interpolateSentenceRanges_exports = {};
|
|
20
20
|
__export(interpolateSentenceRanges_exports, {
|
|
21
|
-
interpolateSentenceRanges: () => interpolateSentenceRanges
|
|
21
|
+
interpolateSentenceRanges: () => interpolateSentenceRanges,
|
|
22
|
+
slotKey: () => slotKey
|
|
22
23
|
});
|
|
23
24
|
module.exports = __toCommonJS(interpolateSentenceRanges_exports);
|
|
24
|
-
function
|
|
25
|
+
function slotKey(slot) {
|
|
26
|
+
return `${slot.chapterId}\0${slot.id}`;
|
|
27
|
+
}
|
|
28
|
+
function slotWeight(slot, sentenceLengths) {
|
|
29
|
+
const length = sentenceLengths[slotKey(slot)];
|
|
30
|
+
return length && length > 0 ? length : 1;
|
|
31
|
+
}
|
|
32
|
+
function buildGapRanges(slots, left, right, audioFileDurations, sentenceLengths) {
|
|
25
33
|
const n = slots.length;
|
|
26
34
|
if (n === 0) return [];
|
|
35
|
+
const weights = slots.map((slot) => slotWeight(slot, sentenceLengths));
|
|
27
36
|
if (left.audiofile === right.audiofile) {
|
|
28
37
|
const span = right.time - left.time;
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
38
|
+
const totalWeight2 = weights.reduce((a, b) => a + b, 0);
|
|
39
|
+
const result2 = [];
|
|
40
|
+
let cursor = left.time;
|
|
41
|
+
for (const [i, slot] of slots.entries()) {
|
|
42
|
+
const start = cursor;
|
|
43
|
+
const end = start + span * weights[i] / totalWeight2;
|
|
44
|
+
result2.push({ ...slot, audiofile: left.audiofile, start, end });
|
|
45
|
+
cursor = end;
|
|
46
|
+
}
|
|
47
|
+
return result2;
|
|
35
48
|
}
|
|
36
49
|
const leftDuration = audioFileDurations[left.audiofile] ?? left.time;
|
|
37
50
|
const leftAvail = leftDuration - left.time;
|
|
38
51
|
const rightAvail = right.time;
|
|
39
52
|
const total = leftAvail + rightAvail;
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
n1 =
|
|
43
|
-
|
|
53
|
+
const totalWeight = weights.reduce((a, b) => a + b, 0);
|
|
54
|
+
const leftWeightShare = total > 0 ? totalWeight * (leftAvail / total) : totalWeight;
|
|
55
|
+
let n1 = 0;
|
|
56
|
+
let accumulatedWeight = 0;
|
|
57
|
+
while (n1 < n && // eslint-disable-next-line @typescript-eslint/no-non-null-assertion
|
|
58
|
+
accumulatedWeight + weights[n1] / 2 <= leftWeightShare) {
|
|
59
|
+
accumulatedWeight += weights[n1];
|
|
60
|
+
n1++;
|
|
61
|
+
}
|
|
62
|
+
const n2 = n - n1;
|
|
44
63
|
const result = [];
|
|
45
64
|
if (n1 > 0) {
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
});
|
|
65
|
+
const leftSlots = slots.slice(0, n1);
|
|
66
|
+
const leftWeights = weights.slice(0, n1);
|
|
67
|
+
const leftTotalWeight = leftWeights.reduce((a, b) => a + b, 0);
|
|
68
|
+
let cursor = left.time;
|
|
69
|
+
for (const [i, slot] of leftSlots.entries()) {
|
|
70
|
+
const start = cursor;
|
|
71
|
+
const end = start + leftAvail * leftWeights[i] / leftTotalWeight;
|
|
72
|
+
result.push({ ...slot, audiofile: left.audiofile, start, end });
|
|
73
|
+
cursor = end;
|
|
54
74
|
}
|
|
55
75
|
}
|
|
56
76
|
if (n2 > 0) {
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
});
|
|
77
|
+
const rightSlots = slots.slice(n1);
|
|
78
|
+
const rightWeights = weights.slice(n1);
|
|
79
|
+
const rightTotalWeight = rightWeights.reduce((a, b) => a + b, 0);
|
|
80
|
+
let cursor = 0;
|
|
81
|
+
for (const [i, slot] of rightSlots.entries()) {
|
|
82
|
+
const start = cursor;
|
|
83
|
+
const end = start + rightAvail * rightWeights[i] / rightTotalWeight;
|
|
84
|
+
result.push({ ...slot, audiofile: right.audiofile, start, end });
|
|
85
|
+
cursor = end;
|
|
65
86
|
}
|
|
66
87
|
}
|
|
67
88
|
return result;
|
|
68
89
|
}
|
|
69
|
-
function interpolateSentenceRanges(sentenceRanges, chapterSentenceCounts, audioFileDurations) {
|
|
90
|
+
function interpolateSentenceRanges(sentenceRanges, chapterSentenceCounts, audioFileDurations, sentenceLengths = {}) {
|
|
70
91
|
if (sentenceRanges.length === 0) return [];
|
|
71
92
|
const result = [];
|
|
72
93
|
const first = sentenceRanges[0];
|
|
@@ -77,7 +98,15 @@ function interpolateSentenceRanges(sentenceRanges, chapterSentenceCounts, audioF
|
|
|
77
98
|
}));
|
|
78
99
|
const left = { time: 0, audiofile: first.audiofile };
|
|
79
100
|
const right = { time: first.start, audiofile: first.audiofile };
|
|
80
|
-
result.push(
|
|
101
|
+
result.push(
|
|
102
|
+
...buildGapRanges(
|
|
103
|
+
slots,
|
|
104
|
+
left,
|
|
105
|
+
right,
|
|
106
|
+
audioFileDurations,
|
|
107
|
+
sentenceLengths
|
|
108
|
+
)
|
|
109
|
+
);
|
|
81
110
|
}
|
|
82
111
|
result.push(first);
|
|
83
112
|
for (let idx = 1; idx < sentenceRanges.length; idx++) {
|
|
@@ -100,7 +129,15 @@ function interpolateSentenceRanges(sentenceRanges, chapterSentenceCounts, audioF
|
|
|
100
129
|
}
|
|
101
130
|
}
|
|
102
131
|
if (gapSlots.length > 0) {
|
|
103
|
-
result.push(
|
|
132
|
+
result.push(
|
|
133
|
+
...buildGapRanges(
|
|
134
|
+
gapSlots,
|
|
135
|
+
left,
|
|
136
|
+
right,
|
|
137
|
+
audioFileDurations,
|
|
138
|
+
sentenceLengths
|
|
139
|
+
)
|
|
140
|
+
);
|
|
104
141
|
}
|
|
105
142
|
result.push(curr);
|
|
106
143
|
}
|
|
@@ -114,11 +151,20 @@ function interpolateSentenceRanges(sentenceRanges, chapterSentenceCounts, audioF
|
|
|
114
151
|
const fileEnd = audioFileDurations[last.audiofile] ?? last.end;
|
|
115
152
|
const left = { time: last.end, audiofile: last.audiofile };
|
|
116
153
|
const right = { time: fileEnd, audiofile: last.audiofile };
|
|
117
|
-
result.push(
|
|
154
|
+
result.push(
|
|
155
|
+
...buildGapRanges(
|
|
156
|
+
slots,
|
|
157
|
+
left,
|
|
158
|
+
right,
|
|
159
|
+
audioFileDurations,
|
|
160
|
+
sentenceLengths
|
|
161
|
+
)
|
|
162
|
+
);
|
|
118
163
|
}
|
|
119
164
|
return result;
|
|
120
165
|
}
|
|
121
166
|
// Annotate the CommonJS export names for ESM import in node:
|
|
122
167
|
0 && (module.exports = {
|
|
123
|
-
interpolateSentenceRanges
|
|
168
|
+
interpolateSentenceRanges,
|
|
169
|
+
slotKey
|
|
124
170
|
});
|
|
@@ -3,6 +3,19 @@ import '@storyteller-platform/ghost-story';
|
|
|
3
3
|
import '@echogarden/text-segmentation';
|
|
4
4
|
import '@storyteller-platform/transliteration';
|
|
5
5
|
|
|
6
|
+
/** A "slot" is a sentence that needs to be synthesised. */
|
|
7
|
+
type Slot = {
|
|
8
|
+
chapterId: string;
|
|
9
|
+
id: number;
|
|
10
|
+
};
|
|
11
|
+
/**
|
|
12
|
+
* Maps a slot to the character length of its sentence text, used to weight
|
|
13
|
+
* how much of a gap's time it should occupy. Keyed by `${chapterId}\0${id}`.
|
|
14
|
+
* Slots missing from this map (or with a non-positive length) fall back to
|
|
15
|
+
* a weight of 1, i.e. even division, same as before this map existed.
|
|
16
|
+
*/
|
|
17
|
+
type SentenceLengths = Record<string, number>;
|
|
18
|
+
declare function slotKey(slot: Slot): string;
|
|
6
19
|
/**
|
|
7
20
|
* Given a sequence of sentence ranges from an entire book,
|
|
8
21
|
* ordered by occurrence in audio, interpolates sentence ranges
|
|
@@ -18,6 +31,6 @@ import '@storyteller-platform/transliteration';
|
|
|
18
31
|
* e.g. chapter001#325 -> chapter002#0, where
|
|
19
32
|
* chapterSentenceCounts["chapter001"] === 330
|
|
20
33
|
*/
|
|
21
|
-
declare function interpolateSentenceRanges(sentenceRanges: SentenceRange[], chapterSentenceCounts: Record<string, number>, audioFileDurations: Record<string, number
|
|
34
|
+
declare function interpolateSentenceRanges(sentenceRanges: SentenceRange[], chapterSentenceCounts: Record<string, number>, audioFileDurations: Record<string, number>, sentenceLengths?: SentenceLengths): SentenceRange[];
|
|
22
35
|
|
|
23
|
-
export { interpolateSentenceRanges };
|
|
36
|
+
export { type SentenceLengths, interpolateSentenceRanges, slotKey };
|
|
@@ -3,6 +3,19 @@ import '@storyteller-platform/ghost-story';
|
|
|
3
3
|
import '@echogarden/text-segmentation';
|
|
4
4
|
import '@storyteller-platform/transliteration';
|
|
5
5
|
|
|
6
|
+
/** A "slot" is a sentence that needs to be synthesised. */
|
|
7
|
+
type Slot = {
|
|
8
|
+
chapterId: string;
|
|
9
|
+
id: number;
|
|
10
|
+
};
|
|
11
|
+
/**
|
|
12
|
+
* Maps a slot to the character length of its sentence text, used to weight
|
|
13
|
+
* how much of a gap's time it should occupy. Keyed by `${chapterId}\0${id}`.
|
|
14
|
+
* Slots missing from this map (or with a non-positive length) fall back to
|
|
15
|
+
* a weight of 1, i.e. even division, same as before this map existed.
|
|
16
|
+
*/
|
|
17
|
+
type SentenceLengths = Record<string, number>;
|
|
18
|
+
declare function slotKey(slot: Slot): string;
|
|
6
19
|
/**
|
|
7
20
|
* Given a sequence of sentence ranges from an entire book,
|
|
8
21
|
* ordered by occurrence in audio, interpolates sentence ranges
|
|
@@ -18,6 +31,6 @@ import '@storyteller-platform/transliteration';
|
|
|
18
31
|
* e.g. chapter001#325 -> chapter002#0, where
|
|
19
32
|
* chapterSentenceCounts["chapter001"] === 330
|
|
20
33
|
*/
|
|
21
|
-
declare function interpolateSentenceRanges(sentenceRanges: SentenceRange[], chapterSentenceCounts: Record<string, number>, audioFileDurations: Record<string, number
|
|
34
|
+
declare function interpolateSentenceRanges(sentenceRanges: SentenceRange[], chapterSentenceCounts: Record<string, number>, audioFileDurations: Record<string, number>, sentenceLengths?: SentenceLengths): SentenceRange[];
|
|
22
35
|
|
|
23
|
-
export { interpolateSentenceRanges };
|
|
36
|
+
export { type SentenceLengths, interpolateSentenceRanges, slotKey };
|
|
@@ -1,50 +1,70 @@
|
|
|
1
1
|
import "../chunk-BIEQXUOY.js";
|
|
2
|
-
function
|
|
2
|
+
function slotKey(slot) {
|
|
3
|
+
return `${slot.chapterId}\0${slot.id}`;
|
|
4
|
+
}
|
|
5
|
+
function slotWeight(slot, sentenceLengths) {
|
|
6
|
+
const length = sentenceLengths[slotKey(slot)];
|
|
7
|
+
return length && length > 0 ? length : 1;
|
|
8
|
+
}
|
|
9
|
+
function buildGapRanges(slots, left, right, audioFileDurations, sentenceLengths) {
|
|
3
10
|
const n = slots.length;
|
|
4
11
|
if (n === 0) return [];
|
|
12
|
+
const weights = slots.map((slot) => slotWeight(slot, sentenceLengths));
|
|
5
13
|
if (left.audiofile === right.audiofile) {
|
|
6
14
|
const span = right.time - left.time;
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
15
|
+
const totalWeight2 = weights.reduce((a, b) => a + b, 0);
|
|
16
|
+
const result2 = [];
|
|
17
|
+
let cursor = left.time;
|
|
18
|
+
for (const [i, slot] of slots.entries()) {
|
|
19
|
+
const start = cursor;
|
|
20
|
+
const end = start + span * weights[i] / totalWeight2;
|
|
21
|
+
result2.push({ ...slot, audiofile: left.audiofile, start, end });
|
|
22
|
+
cursor = end;
|
|
23
|
+
}
|
|
24
|
+
return result2;
|
|
13
25
|
}
|
|
14
26
|
const leftDuration = audioFileDurations[left.audiofile] ?? left.time;
|
|
15
27
|
const leftAvail = leftDuration - left.time;
|
|
16
28
|
const rightAvail = right.time;
|
|
17
29
|
const total = leftAvail + rightAvail;
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
n1 =
|
|
21
|
-
|
|
30
|
+
const totalWeight = weights.reduce((a, b) => a + b, 0);
|
|
31
|
+
const leftWeightShare = total > 0 ? totalWeight * (leftAvail / total) : totalWeight;
|
|
32
|
+
let n1 = 0;
|
|
33
|
+
let accumulatedWeight = 0;
|
|
34
|
+
while (n1 < n && // eslint-disable-next-line @typescript-eslint/no-non-null-assertion
|
|
35
|
+
accumulatedWeight + weights[n1] / 2 <= leftWeightShare) {
|
|
36
|
+
accumulatedWeight += weights[n1];
|
|
37
|
+
n1++;
|
|
38
|
+
}
|
|
39
|
+
const n2 = n - n1;
|
|
22
40
|
const result = [];
|
|
23
41
|
if (n1 > 0) {
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
});
|
|
42
|
+
const leftSlots = slots.slice(0, n1);
|
|
43
|
+
const leftWeights = weights.slice(0, n1);
|
|
44
|
+
const leftTotalWeight = leftWeights.reduce((a, b) => a + b, 0);
|
|
45
|
+
let cursor = left.time;
|
|
46
|
+
for (const [i, slot] of leftSlots.entries()) {
|
|
47
|
+
const start = cursor;
|
|
48
|
+
const end = start + leftAvail * leftWeights[i] / leftTotalWeight;
|
|
49
|
+
result.push({ ...slot, audiofile: left.audiofile, start, end });
|
|
50
|
+
cursor = end;
|
|
32
51
|
}
|
|
33
52
|
}
|
|
34
53
|
if (n2 > 0) {
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
});
|
|
54
|
+
const rightSlots = slots.slice(n1);
|
|
55
|
+
const rightWeights = weights.slice(n1);
|
|
56
|
+
const rightTotalWeight = rightWeights.reduce((a, b) => a + b, 0);
|
|
57
|
+
let cursor = 0;
|
|
58
|
+
for (const [i, slot] of rightSlots.entries()) {
|
|
59
|
+
const start = cursor;
|
|
60
|
+
const end = start + rightAvail * rightWeights[i] / rightTotalWeight;
|
|
61
|
+
result.push({ ...slot, audiofile: right.audiofile, start, end });
|
|
62
|
+
cursor = end;
|
|
43
63
|
}
|
|
44
64
|
}
|
|
45
65
|
return result;
|
|
46
66
|
}
|
|
47
|
-
function interpolateSentenceRanges(sentenceRanges, chapterSentenceCounts, audioFileDurations) {
|
|
67
|
+
function interpolateSentenceRanges(sentenceRanges, chapterSentenceCounts, audioFileDurations, sentenceLengths = {}) {
|
|
48
68
|
if (sentenceRanges.length === 0) return [];
|
|
49
69
|
const result = [];
|
|
50
70
|
const first = sentenceRanges[0];
|
|
@@ -55,7 +75,15 @@ function interpolateSentenceRanges(sentenceRanges, chapterSentenceCounts, audioF
|
|
|
55
75
|
}));
|
|
56
76
|
const left = { time: 0, audiofile: first.audiofile };
|
|
57
77
|
const right = { time: first.start, audiofile: first.audiofile };
|
|
58
|
-
result.push(
|
|
78
|
+
result.push(
|
|
79
|
+
...buildGapRanges(
|
|
80
|
+
slots,
|
|
81
|
+
left,
|
|
82
|
+
right,
|
|
83
|
+
audioFileDurations,
|
|
84
|
+
sentenceLengths
|
|
85
|
+
)
|
|
86
|
+
);
|
|
59
87
|
}
|
|
60
88
|
result.push(first);
|
|
61
89
|
for (let idx = 1; idx < sentenceRanges.length; idx++) {
|
|
@@ -78,7 +106,15 @@ function interpolateSentenceRanges(sentenceRanges, chapterSentenceCounts, audioF
|
|
|
78
106
|
}
|
|
79
107
|
}
|
|
80
108
|
if (gapSlots.length > 0) {
|
|
81
|
-
result.push(
|
|
109
|
+
result.push(
|
|
110
|
+
...buildGapRanges(
|
|
111
|
+
gapSlots,
|
|
112
|
+
left,
|
|
113
|
+
right,
|
|
114
|
+
audioFileDurations,
|
|
115
|
+
sentenceLengths
|
|
116
|
+
)
|
|
117
|
+
);
|
|
82
118
|
}
|
|
83
119
|
result.push(curr);
|
|
84
120
|
}
|
|
@@ -92,10 +128,19 @@ function interpolateSentenceRanges(sentenceRanges, chapterSentenceCounts, audioF
|
|
|
92
128
|
const fileEnd = audioFileDurations[last.audiofile] ?? last.end;
|
|
93
129
|
const left = { time: last.end, audiofile: last.audiofile };
|
|
94
130
|
const right = { time: fileEnd, audiofile: last.audiofile };
|
|
95
|
-
result.push(
|
|
131
|
+
result.push(
|
|
132
|
+
...buildGapRanges(
|
|
133
|
+
slots,
|
|
134
|
+
left,
|
|
135
|
+
right,
|
|
136
|
+
audioFileDurations,
|
|
137
|
+
sentenceLengths
|
|
138
|
+
)
|
|
139
|
+
);
|
|
96
140
|
}
|
|
97
141
|
return result;
|
|
98
142
|
}
|
|
99
143
|
export {
|
|
100
|
-
interpolateSentenceRanges
|
|
144
|
+
interpolateSentenceRanges,
|
|
145
|
+
slotKey
|
|
101
146
|
};
|
package/dist/common/ffmpeg.cjs
CHANGED
|
@@ -28,8 +28,11 @@ var __toESM = (mod, isNodeMode, target) => (target = mod != null ? __create(__ge
|
|
|
28
28
|
var __toCommonJS = (mod) => __copyProps(__defProp({}, "__esModule", { value: true }), mod);
|
|
29
29
|
var ffmpeg_exports = {};
|
|
30
30
|
__export(ffmpeg_exports, {
|
|
31
|
+
MP3_CBR_BITRATES: () => MP3_CBR_BITRATES,
|
|
31
32
|
getTrackDuration: () => getTrackDuration,
|
|
32
33
|
getTrackInfo: () => getTrackInfo,
|
|
34
|
+
isVbrMp3: () => isVbrMp3,
|
|
35
|
+
selectCbrBitrate: () => selectCbrBitrate,
|
|
33
36
|
splitFile: () => splitFile,
|
|
34
37
|
transcodeFile: () => transcodeFile
|
|
35
38
|
});
|
|
@@ -74,6 +77,52 @@ async function getTrackDuration(path, logger) {
|
|
|
74
77
|
const info = await getTrackInfo(path, logger);
|
|
75
78
|
return info["duration"];
|
|
76
79
|
}
|
|
80
|
+
const MP3_CBR_BITRATES = [
|
|
81
|
+
64e3,
|
|
82
|
+
8e4,
|
|
83
|
+
96e3,
|
|
84
|
+
112e3,
|
|
85
|
+
128e3,
|
|
86
|
+
16e4,
|
|
87
|
+
192e3,
|
|
88
|
+
224e3,
|
|
89
|
+
256e3,
|
|
90
|
+
32e4
|
|
91
|
+
];
|
|
92
|
+
const VBR_PROBE_PACKET_COUNT = 50;
|
|
93
|
+
const MP3_CBR_MAX_DISTINCT_SIZES = 2;
|
|
94
|
+
const VBR_PROBE_MIN_SEEKABLE_SECONDS = 180;
|
|
95
|
+
async function probeAudioDuration(path) {
|
|
96
|
+
const stdout = await execCmd(
|
|
97
|
+
`ffprobe -i ${(0, import_shell.quotePath)(path)} -v error -show_entries format=duration -output_format json`
|
|
98
|
+
);
|
|
99
|
+
const { format } = JSON.parse(stdout);
|
|
100
|
+
const duration = Number(format?.duration);
|
|
101
|
+
return Number.isFinite(duration) && duration > 0 ? duration : null;
|
|
102
|
+
}
|
|
103
|
+
async function probePacketSizes(path, startSeconds) {
|
|
104
|
+
const interval = startSeconds > 0 ? `${startSeconds}%+#${VBR_PROBE_PACKET_COUNT}` : `%+#${VBR_PROBE_PACKET_COUNT}`;
|
|
105
|
+
const stdout = await execCmd(
|
|
106
|
+
`ffprobe -i ${(0, import_shell.quotePath)(path)} -v error -select_streams a:0 -read_intervals "${interval}" -show_entries packet=size -output_format json`
|
|
107
|
+
);
|
|
108
|
+
const { packets } = JSON.parse(stdout);
|
|
109
|
+
return (packets ?? []).map((packet) => Number(packet.size)).filter((size) => Number.isFinite(size) && size > 0);
|
|
110
|
+
}
|
|
111
|
+
async function isVbrMp3(path) {
|
|
112
|
+
if ((0, import_node_path.extname)(path).toLowerCase() !== ".mp3") return false;
|
|
113
|
+
const duration = await probeAudioDuration(path);
|
|
114
|
+
const startSeconds = duration && duration > VBR_PROBE_MIN_SEEKABLE_SECONDS ? Math.floor(duration / 3) : 0;
|
|
115
|
+
let sizes = await probePacketSizes(path, startSeconds);
|
|
116
|
+
if (sizes.length === 0 && startSeconds > 0) {
|
|
117
|
+
sizes = await probePacketSizes(path, 0);
|
|
118
|
+
}
|
|
119
|
+
if (sizes.length === 0) return false;
|
|
120
|
+
const distinctSizes = new Set(sizes).size;
|
|
121
|
+
return distinctSizes > MP3_CBR_MAX_DISTINCT_SIZES;
|
|
122
|
+
}
|
|
123
|
+
function selectCbrBitrate(averageBitrate) {
|
|
124
|
+
return MP3_CBR_BITRATES.find((tier) => tier >= averageBitrate) ?? MP3_CBR_BITRATES.at(-1) ?? MP3_CBR_BITRATES[0];
|
|
125
|
+
}
|
|
77
126
|
function parseTrackInfo(format) {
|
|
78
127
|
return {
|
|
79
128
|
filename: format.filename,
|
|
@@ -136,15 +185,16 @@ async function constructExtractCoverArtCommand(source, destExtension) {
|
|
|
136
185
|
];
|
|
137
186
|
return `${command} ${args.join(" ")} | `;
|
|
138
187
|
}
|
|
139
|
-
function commonFfmpegArguments(
|
|
188
|
+
function commonFfmpegArguments(options) {
|
|
189
|
+
const { sourceExtension, destExtension, codec, bitrate } = options;
|
|
140
190
|
const args = ["-vn"];
|
|
141
191
|
if (codec) {
|
|
142
|
-
args.push(
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
192
|
+
args.push("-c:a", codec);
|
|
193
|
+
if (codec === "libopus") {
|
|
194
|
+
args.push("-b:a", bitrate && /^\d+[kK]$/i.test(bitrate) ? bitrate : "32K");
|
|
195
|
+
} else if (codec === "libmp3lame" && bitrate) {
|
|
196
|
+
args.push("-b:a", bitrate);
|
|
197
|
+
}
|
|
148
198
|
} else if ((0, import_mime.areSameType)(sourceExtension, destExtension) || destExtension == ".mp4") {
|
|
149
199
|
args.push("-c:a", "copy");
|
|
150
200
|
}
|
|
@@ -168,12 +218,12 @@ async function splitFile(input, output, start, end, encoding, signal, logger) {
|
|
|
168
218
|
end,
|
|
169
219
|
"-i",
|
|
170
220
|
(0, import_shell.quotePath)(input),
|
|
171
|
-
...commonFfmpegArguments(
|
|
172
|
-
(0, import_node_path.extname)(input),
|
|
173
|
-
(0, import_node_path.extname)(output),
|
|
174
|
-
encoding?.codec ?? null,
|
|
175
|
-
encoding?.bitrate ?? null
|
|
176
|
-
),
|
|
221
|
+
...commonFfmpegArguments({
|
|
222
|
+
sourceExtension: (0, import_node_path.extname)(input),
|
|
223
|
+
destExtension: (0, import_node_path.extname)(output),
|
|
224
|
+
codec: encoding?.codec ?? null,
|
|
225
|
+
bitrate: encoding?.bitrate ?? null
|
|
226
|
+
}),
|
|
177
227
|
(0, import_shell.quotePath)(output)
|
|
178
228
|
];
|
|
179
229
|
const coverArtCommand = await constructExtractCoverArtCommand(
|
|
@@ -203,12 +253,12 @@ async function transcodeFile(input, output, encoding, signal, logger) {
|
|
|
203
253
|
"-nostdin",
|
|
204
254
|
"-i",
|
|
205
255
|
(0, import_shell.quotePath)(input),
|
|
206
|
-
...commonFfmpegArguments(
|
|
207
|
-
(0, import_node_path.extname)(input),
|
|
208
|
-
(0, import_node_path.extname)(output),
|
|
209
|
-
encoding?.codec ?? null,
|
|
210
|
-
encoding?.bitrate ?? null
|
|
211
|
-
),
|
|
256
|
+
...commonFfmpegArguments({
|
|
257
|
+
sourceExtension: (0, import_node_path.extname)(input),
|
|
258
|
+
destExtension: (0, import_node_path.extname)(output),
|
|
259
|
+
codec: encoding?.codec ?? null,
|
|
260
|
+
bitrate: encoding?.bitrate ?? null
|
|
261
|
+
}),
|
|
212
262
|
(0, import_shell.quotePath)(output)
|
|
213
263
|
];
|
|
214
264
|
const coverArtCommand = await constructExtractCoverArtCommand(
|
|
@@ -224,8 +274,11 @@ async function transcodeFile(input, output, encoding, signal, logger) {
|
|
|
224
274
|
}
|
|
225
275
|
// Annotate the CommonJS export names for ESM import in node:
|
|
226
276
|
0 && (module.exports = {
|
|
277
|
+
MP3_CBR_BITRATES,
|
|
227
278
|
getTrackDuration,
|
|
228
279
|
getTrackInfo,
|
|
280
|
+
isVbrMp3,
|
|
281
|
+
selectCbrBitrate,
|
|
229
282
|
splitFile,
|
|
230
283
|
transcodeFile
|
|
231
284
|
});
|
package/dist/common/ffmpeg.d.cts
CHANGED
|
@@ -3,6 +3,22 @@ import { AudioEncoding } from '../process/AudioEncoding.cjs';
|
|
|
3
3
|
|
|
4
4
|
declare const getTrackInfo: (path: string, logger?: Logger) => Promise<TrackInfo>;
|
|
5
5
|
declare function getTrackDuration(path: string, logger?: Logger): Promise<number>;
|
|
6
|
+
/**
|
|
7
|
+
* CBR bitrates (bps) offered for MP3 output, roughly matching LAME -V9..-V0
|
|
8
|
+
*/
|
|
9
|
+
declare const MP3_CBR_BITRATES: readonly [64000, 80000, 96000, 112000, 128000, 160000, 192000, 224000, 256000, 320000];
|
|
10
|
+
/**
|
|
11
|
+
* Detect whether an MP3 file uses a variable bitrate
|
|
12
|
+
* Does this by sampling the first few packets and checking if the sizes are different
|
|
13
|
+
* CBR MP3 files will have the same packet size for the entire file
|
|
14
|
+
*
|
|
15
|
+
* Can't really trust the reported bitrate to tell CBR from VBR
|
|
16
|
+
* LAME writes a Xing header carrying the *average* bitrate,
|
|
17
|
+
* which ffprobe surfaces as a normal per-stream `bit_rate`,
|
|
18
|
+
* so a VBR file looks identical to a CBR one by that measure.
|
|
19
|
+
*/
|
|
20
|
+
declare function isVbrMp3(path: string): Promise<boolean>;
|
|
21
|
+
declare function selectCbrBitrate(averageBitrate: number): number;
|
|
6
22
|
type TrackInfo = {
|
|
7
23
|
filename: string;
|
|
8
24
|
nbStreams: number;
|
|
@@ -30,4 +46,4 @@ type TrackInfo = {
|
|
|
30
46
|
declare function splitFile(input: string, output: string, start: number, end: number, encoding?: AudioEncoding | null, signal?: AbortSignal | null, logger?: Logger | null): Promise<boolean>;
|
|
31
47
|
declare function transcodeFile(input: string, output: string, encoding?: AudioEncoding | null, signal?: AbortSignal | null, logger?: Logger | null): Promise<true | undefined>;
|
|
32
48
|
|
|
33
|
-
export { getTrackDuration, getTrackInfo, splitFile, transcodeFile };
|
|
49
|
+
export { MP3_CBR_BITRATES, getTrackDuration, getTrackInfo, isVbrMp3, selectCbrBitrate, splitFile, transcodeFile };
|
package/dist/common/ffmpeg.d.ts
CHANGED
|
@@ -3,6 +3,22 @@ import { AudioEncoding } from '../process/AudioEncoding.js';
|
|
|
3
3
|
|
|
4
4
|
declare const getTrackInfo: (path: string, logger?: Logger) => Promise<TrackInfo>;
|
|
5
5
|
declare function getTrackDuration(path: string, logger?: Logger): Promise<number>;
|
|
6
|
+
/**
|
|
7
|
+
* CBR bitrates (bps) offered for MP3 output, roughly matching LAME -V9..-V0
|
|
8
|
+
*/
|
|
9
|
+
declare const MP3_CBR_BITRATES: readonly [64000, 80000, 96000, 112000, 128000, 160000, 192000, 224000, 256000, 320000];
|
|
10
|
+
/**
|
|
11
|
+
* Detect whether an MP3 file uses a variable bitrate
|
|
12
|
+
* Does this by sampling the first few packets and checking if the sizes are different
|
|
13
|
+
* CBR MP3 files will have the same packet size for the entire file
|
|
14
|
+
*
|
|
15
|
+
* Can't really trust the reported bitrate to tell CBR from VBR
|
|
16
|
+
* LAME writes a Xing header carrying the *average* bitrate,
|
|
17
|
+
* which ffprobe surfaces as a normal per-stream `bit_rate`,
|
|
18
|
+
* so a VBR file looks identical to a CBR one by that measure.
|
|
19
|
+
*/
|
|
20
|
+
declare function isVbrMp3(path: string): Promise<boolean>;
|
|
21
|
+
declare function selectCbrBitrate(averageBitrate: number): number;
|
|
6
22
|
type TrackInfo = {
|
|
7
23
|
filename: string;
|
|
8
24
|
nbStreams: number;
|
|
@@ -30,4 +46,4 @@ type TrackInfo = {
|
|
|
30
46
|
declare function splitFile(input: string, output: string, start: number, end: number, encoding?: AudioEncoding | null, signal?: AbortSignal | null, logger?: Logger | null): Promise<boolean>;
|
|
31
47
|
declare function transcodeFile(input: string, output: string, encoding?: AudioEncoding | null, signal?: AbortSignal | null, logger?: Logger | null): Promise<true | undefined>;
|
|
32
48
|
|
|
33
|
-
export { getTrackDuration, getTrackInfo, splitFile, transcodeFile };
|
|
49
|
+
export { MP3_CBR_BITRATES, getTrackDuration, getTrackInfo, isVbrMp3, selectCbrBitrate, splitFile, transcodeFile };
|
package/dist/common/ffmpeg.js
CHANGED
|
@@ -39,6 +39,52 @@ async function getTrackDuration(path, logger) {
|
|
|
39
39
|
const info = await getTrackInfo(path, logger);
|
|
40
40
|
return info["duration"];
|
|
41
41
|
}
|
|
42
|
+
const MP3_CBR_BITRATES = [
|
|
43
|
+
64e3,
|
|
44
|
+
8e4,
|
|
45
|
+
96e3,
|
|
46
|
+
112e3,
|
|
47
|
+
128e3,
|
|
48
|
+
16e4,
|
|
49
|
+
192e3,
|
|
50
|
+
224e3,
|
|
51
|
+
256e3,
|
|
52
|
+
32e4
|
|
53
|
+
];
|
|
54
|
+
const VBR_PROBE_PACKET_COUNT = 50;
|
|
55
|
+
const MP3_CBR_MAX_DISTINCT_SIZES = 2;
|
|
56
|
+
const VBR_PROBE_MIN_SEEKABLE_SECONDS = 180;
|
|
57
|
+
async function probeAudioDuration(path) {
|
|
58
|
+
const stdout = await execCmd(
|
|
59
|
+
`ffprobe -i ${quotePath(path)} -v error -show_entries format=duration -output_format json`
|
|
60
|
+
);
|
|
61
|
+
const { format } = JSON.parse(stdout);
|
|
62
|
+
const duration = Number(format?.duration);
|
|
63
|
+
return Number.isFinite(duration) && duration > 0 ? duration : null;
|
|
64
|
+
}
|
|
65
|
+
async function probePacketSizes(path, startSeconds) {
|
|
66
|
+
const interval = startSeconds > 0 ? `${startSeconds}%+#${VBR_PROBE_PACKET_COUNT}` : `%+#${VBR_PROBE_PACKET_COUNT}`;
|
|
67
|
+
const stdout = await execCmd(
|
|
68
|
+
`ffprobe -i ${quotePath(path)} -v error -select_streams a:0 -read_intervals "${interval}" -show_entries packet=size -output_format json`
|
|
69
|
+
);
|
|
70
|
+
const { packets } = JSON.parse(stdout);
|
|
71
|
+
return (packets ?? []).map((packet) => Number(packet.size)).filter((size) => Number.isFinite(size) && size > 0);
|
|
72
|
+
}
|
|
73
|
+
async function isVbrMp3(path) {
|
|
74
|
+
if (extname(path).toLowerCase() !== ".mp3") return false;
|
|
75
|
+
const duration = await probeAudioDuration(path);
|
|
76
|
+
const startSeconds = duration && duration > VBR_PROBE_MIN_SEEKABLE_SECONDS ? Math.floor(duration / 3) : 0;
|
|
77
|
+
let sizes = await probePacketSizes(path, startSeconds);
|
|
78
|
+
if (sizes.length === 0 && startSeconds > 0) {
|
|
79
|
+
sizes = await probePacketSizes(path, 0);
|
|
80
|
+
}
|
|
81
|
+
if (sizes.length === 0) return false;
|
|
82
|
+
const distinctSizes = new Set(sizes).size;
|
|
83
|
+
return distinctSizes > MP3_CBR_MAX_DISTINCT_SIZES;
|
|
84
|
+
}
|
|
85
|
+
function selectCbrBitrate(averageBitrate) {
|
|
86
|
+
return MP3_CBR_BITRATES.find((tier) => tier >= averageBitrate) ?? MP3_CBR_BITRATES.at(-1) ?? MP3_CBR_BITRATES[0];
|
|
87
|
+
}
|
|
42
88
|
function parseTrackInfo(format) {
|
|
43
89
|
return {
|
|
44
90
|
filename: format.filename,
|
|
@@ -101,15 +147,16 @@ async function constructExtractCoverArtCommand(source, destExtension) {
|
|
|
101
147
|
];
|
|
102
148
|
return `${command} ${args.join(" ")} | `;
|
|
103
149
|
}
|
|
104
|
-
function commonFfmpegArguments(
|
|
150
|
+
function commonFfmpegArguments(options) {
|
|
151
|
+
const { sourceExtension, destExtension, codec, bitrate } = options;
|
|
105
152
|
const args = ["-vn"];
|
|
106
153
|
if (codec) {
|
|
107
|
-
args.push(
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
154
|
+
args.push("-c:a", codec);
|
|
155
|
+
if (codec === "libopus") {
|
|
156
|
+
args.push("-b:a", bitrate && /^\d+[kK]$/i.test(bitrate) ? bitrate : "32K");
|
|
157
|
+
} else if (codec === "libmp3lame" && bitrate) {
|
|
158
|
+
args.push("-b:a", bitrate);
|
|
159
|
+
}
|
|
113
160
|
} else if (areSameType(sourceExtension, destExtension) || destExtension == ".mp4") {
|
|
114
161
|
args.push("-c:a", "copy");
|
|
115
162
|
}
|
|
@@ -133,12 +180,12 @@ async function splitFile(input, output, start, end, encoding, signal, logger) {
|
|
|
133
180
|
end,
|
|
134
181
|
"-i",
|
|
135
182
|
quotePath(input),
|
|
136
|
-
...commonFfmpegArguments(
|
|
137
|
-
extname(input),
|
|
138
|
-
extname(output),
|
|
139
|
-
encoding?.codec ?? null,
|
|
140
|
-
encoding?.bitrate ?? null
|
|
141
|
-
),
|
|
183
|
+
...commonFfmpegArguments({
|
|
184
|
+
sourceExtension: extname(input),
|
|
185
|
+
destExtension: extname(output),
|
|
186
|
+
codec: encoding?.codec ?? null,
|
|
187
|
+
bitrate: encoding?.bitrate ?? null
|
|
188
|
+
}),
|
|
142
189
|
quotePath(output)
|
|
143
190
|
];
|
|
144
191
|
const coverArtCommand = await constructExtractCoverArtCommand(
|
|
@@ -168,12 +215,12 @@ async function transcodeFile(input, output, encoding, signal, logger) {
|
|
|
168
215
|
"-nostdin",
|
|
169
216
|
"-i",
|
|
170
217
|
quotePath(input),
|
|
171
|
-
...commonFfmpegArguments(
|
|
172
|
-
extname(input),
|
|
173
|
-
extname(output),
|
|
174
|
-
encoding?.codec ?? null,
|
|
175
|
-
encoding?.bitrate ?? null
|
|
176
|
-
),
|
|
218
|
+
...commonFfmpegArguments({
|
|
219
|
+
sourceExtension: extname(input),
|
|
220
|
+
destExtension: extname(output),
|
|
221
|
+
codec: encoding?.codec ?? null,
|
|
222
|
+
bitrate: encoding?.bitrate ?? null
|
|
223
|
+
}),
|
|
177
224
|
quotePath(output)
|
|
178
225
|
];
|
|
179
226
|
const coverArtCommand = await constructExtractCoverArtCommand(
|
|
@@ -188,8 +235,11 @@ async function transcodeFile(input, output, encoding, signal, logger) {
|
|
|
188
235
|
return true;
|
|
189
236
|
}
|
|
190
237
|
export {
|
|
238
|
+
MP3_CBR_BITRATES,
|
|
191
239
|
getTrackDuration,
|
|
192
240
|
getTrackInfo,
|
|
241
|
+
isVbrMp3,
|
|
242
|
+
selectCbrBitrate,
|
|
193
243
|
splitFile,
|
|
194
244
|
transcodeFile
|
|
195
245
|
};
|
|
@@ -126,6 +126,27 @@ async function processAudiobook(input, output, options) {
|
|
|
126
126
|
);
|
|
127
127
|
return timing;
|
|
128
128
|
}
|
|
129
|
+
async function resolveVbrEncoding(filepath, userEncoding, logger) {
|
|
130
|
+
if (userEncoding?.codec && userEncoding.codec !== "libmp3lame") {
|
|
131
|
+
return userEncoding;
|
|
132
|
+
}
|
|
133
|
+
const sourceIsMp3 = (0, import_node_path.extname)(filepath).toLowerCase() === ".mp3";
|
|
134
|
+
if (!userEncoding?.codec && !sourceIsMp3) {
|
|
135
|
+
return userEncoding;
|
|
136
|
+
}
|
|
137
|
+
if (!userEncoding?.codec && !await (0, import_ffmpeg.isVbrMp3)(filepath)) {
|
|
138
|
+
return userEncoding;
|
|
139
|
+
}
|
|
140
|
+
const trackInfo = await (0, import_ffmpeg.getTrackInfo)(filepath, logger ?? void 0);
|
|
141
|
+
const targetBitrate = (0, import_ffmpeg.selectCbrBitrate)(trackInfo.bitRate);
|
|
142
|
+
logger?.info(
|
|
143
|
+
`Forcing CBR MP3 for ${filepath} (avg ${trackInfo.bitRate}bps) at ${targetBitrate / 1e3}k`
|
|
144
|
+
);
|
|
145
|
+
return {
|
|
146
|
+
codec: "libmp3lame",
|
|
147
|
+
bitrate: `${targetBitrate / 1e3}k`
|
|
148
|
+
};
|
|
149
|
+
}
|
|
129
150
|
async function processFile(input, output, prefix, options) {
|
|
130
151
|
var _stack = [];
|
|
131
152
|
try {
|
|
@@ -144,13 +165,26 @@ async function processFile(input, output, prefix, options) {
|
|
|
144
165
|
options.signal,
|
|
145
166
|
options.logger
|
|
146
167
|
);
|
|
168
|
+
const vbrEncodings = /* @__PURE__ */ new Map();
|
|
169
|
+
const uniqueFilepaths = [...new Set(ranges.map((r) => r.filepath))];
|
|
170
|
+
await Promise.all(
|
|
171
|
+
uniqueFilepaths.map(async (filepath) => {
|
|
172
|
+
const result = await resolveVbrEncoding(
|
|
173
|
+
filepath,
|
|
174
|
+
options.encoding,
|
|
175
|
+
options.logger
|
|
176
|
+
);
|
|
177
|
+
vbrEncodings.set(filepath, result);
|
|
178
|
+
})
|
|
179
|
+
);
|
|
147
180
|
await Promise.all(
|
|
148
181
|
ranges.map(async (range, index) => {
|
|
149
182
|
var _stack2 = [];
|
|
150
183
|
try {
|
|
184
|
+
const effectiveEncoding = vbrEncodings.has(range.filepath) ? vbrEncodings.get(range.filepath) : options.encoding;
|
|
151
185
|
const outputExtension = determineExtension(
|
|
152
186
|
range.filepath,
|
|
153
|
-
|
|
187
|
+
effectiveEncoding?.codec
|
|
154
188
|
);
|
|
155
189
|
const outputFilename = `${prefix}${(index + 1).toString().padStart(5, "0")}${outputExtension}`;
|
|
156
190
|
const outputFilepath = (0, import_node_path.join)(output, outputFilename);
|
|
@@ -168,7 +202,7 @@ async function processFile(input, output, prefix, options) {
|
|
|
168
202
|
outputFilepath,
|
|
169
203
|
range.start,
|
|
170
204
|
range.end,
|
|
171
|
-
|
|
205
|
+
effectiveEncoding,
|
|
172
206
|
options.signal,
|
|
173
207
|
options.logger
|
|
174
208
|
);
|
|
@@ -19,7 +19,12 @@ import {
|
|
|
19
19
|
createAggregator,
|
|
20
20
|
createTiming
|
|
21
21
|
} from "@storyteller-platform/ghost-story";
|
|
22
|
-
import {
|
|
22
|
+
import {
|
|
23
|
+
getTrackInfo,
|
|
24
|
+
isVbrMp3,
|
|
25
|
+
selectCbrBitrate,
|
|
26
|
+
splitFile
|
|
27
|
+
} from "../common/ffmpeg.js";
|
|
23
28
|
import { getSafeChapterRanges } from "./ranges.js";
|
|
24
29
|
async function processAudiobook(input, output, options) {
|
|
25
30
|
const timing = createAggregator();
|
|
@@ -73,6 +78,27 @@ async function processAudiobook(input, output, options) {
|
|
|
73
78
|
);
|
|
74
79
|
return timing;
|
|
75
80
|
}
|
|
81
|
+
async function resolveVbrEncoding(filepath, userEncoding, logger) {
|
|
82
|
+
if (userEncoding?.codec && userEncoding.codec !== "libmp3lame") {
|
|
83
|
+
return userEncoding;
|
|
84
|
+
}
|
|
85
|
+
const sourceIsMp3 = extname(filepath).toLowerCase() === ".mp3";
|
|
86
|
+
if (!userEncoding?.codec && !sourceIsMp3) {
|
|
87
|
+
return userEncoding;
|
|
88
|
+
}
|
|
89
|
+
if (!userEncoding?.codec && !await isVbrMp3(filepath)) {
|
|
90
|
+
return userEncoding;
|
|
91
|
+
}
|
|
92
|
+
const trackInfo = await getTrackInfo(filepath, logger ?? void 0);
|
|
93
|
+
const targetBitrate = selectCbrBitrate(trackInfo.bitRate);
|
|
94
|
+
logger?.info(
|
|
95
|
+
`Forcing CBR MP3 for ${filepath} (avg ${trackInfo.bitRate}bps) at ${targetBitrate / 1e3}k`
|
|
96
|
+
);
|
|
97
|
+
return {
|
|
98
|
+
codec: "libmp3lame",
|
|
99
|
+
bitrate: `${targetBitrate / 1e3}k`
|
|
100
|
+
};
|
|
101
|
+
}
|
|
76
102
|
async function processFile(input, output, prefix, options) {
|
|
77
103
|
var _stack = [];
|
|
78
104
|
try {
|
|
@@ -91,13 +117,26 @@ async function processFile(input, output, prefix, options) {
|
|
|
91
117
|
options.signal,
|
|
92
118
|
options.logger
|
|
93
119
|
);
|
|
120
|
+
const vbrEncodings = /* @__PURE__ */ new Map();
|
|
121
|
+
const uniqueFilepaths = [...new Set(ranges.map((r) => r.filepath))];
|
|
122
|
+
await Promise.all(
|
|
123
|
+
uniqueFilepaths.map(async (filepath) => {
|
|
124
|
+
const result = await resolveVbrEncoding(
|
|
125
|
+
filepath,
|
|
126
|
+
options.encoding,
|
|
127
|
+
options.logger
|
|
128
|
+
);
|
|
129
|
+
vbrEncodings.set(filepath, result);
|
|
130
|
+
})
|
|
131
|
+
);
|
|
94
132
|
await Promise.all(
|
|
95
133
|
ranges.map(async (range, index) => {
|
|
96
134
|
var _stack2 = [];
|
|
97
135
|
try {
|
|
136
|
+
const effectiveEncoding = vbrEncodings.has(range.filepath) ? vbrEncodings.get(range.filepath) : options.encoding;
|
|
98
137
|
const outputExtension = determineExtension(
|
|
99
138
|
range.filepath,
|
|
100
|
-
|
|
139
|
+
effectiveEncoding?.codec
|
|
101
140
|
);
|
|
102
141
|
const outputFilename = `${prefix}${(index + 1).toString().padStart(5, "0")}${outputExtension}`;
|
|
103
142
|
const outputFilepath = join(output, outputFilename);
|
|
@@ -115,7 +154,7 @@ async function processFile(input, output, prefix, options) {
|
|
|
115
154
|
outputFilepath,
|
|
116
155
|
range.start,
|
|
117
156
|
range.end,
|
|
118
|
-
|
|
157
|
+
effectiveEncoding,
|
|
119
158
|
options.signal,
|
|
120
159
|
options.logger
|
|
121
160
|
);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@storyteller-platform/align",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.49",
|
|
4
4
|
"description": "A library and CLI for automatically aligning audiobooks and EPUBs to produce Media Overlays",
|
|
5
5
|
"author": "Shane Friedman",
|
|
6
6
|
"license": "MIT",
|
|
@@ -71,7 +71,7 @@
|
|
|
71
71
|
"@optique/run": "^0.10.7",
|
|
72
72
|
"@readium/shared": "^2.2.0",
|
|
73
73
|
"@storyteller-platform/audiobook": "^0.4.1",
|
|
74
|
-
"@storyteller-platform/epub": "^0.6.
|
|
74
|
+
"@storyteller-platform/epub": "^0.6.2",
|
|
75
75
|
"@storyteller-platform/ghost-story": "^0.1.11",
|
|
76
76
|
"@storyteller-platform/transliteration": "^3.1.2",
|
|
77
77
|
"chalk": "^5.4.1",
|