@mengine/medeo-client 2.0.1-alpha.7 → 2.0.1-alpha.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{document-Bnxm1p5E.js → document-C30rnewk.js} +136 -32
- package/dist/index.d.ts +30 -10
- package/dist/index.js +102 -61
- package/dist/testing.js +1 -1
- package/package.json +6 -6
|
@@ -1447,8 +1447,85 @@ function emptyLaneTracks() {
|
|
|
1447
1447
|
}
|
|
1448
1448
|
//#endregion
|
|
1449
1449
|
//#region ../medeo-dsl/src/entities.ts
|
|
1450
|
+
/**
|
|
1451
|
+
* What every known entity kind declares.
|
|
1452
|
+
*
|
|
1453
|
+
* `kinds` is a multi-declaration: every entry holds at the same time, the first
|
|
1454
|
+
* naming the object and the rest the properties it shares with others. The
|
|
1455
|
+
* order carries no priority, endpoint role, or edit sequence — an operation
|
|
1456
|
+
* asks whether the entity declares the kind it needs, never what its "main
|
|
1457
|
+
* type" is.
|
|
1458
|
+
*
|
|
1459
|
+
* `Caption` declares `Sequence` on purpose: it owns the display timing of the
|
|
1460
|
+
* one AudioScript segment it shows, which is what admits it into a Clip.
|
|
1461
|
+
* `Voice` declares neither `MediaFile` nor `Sequence` — it is a timbre
|
|
1462
|
+
* identity, not playable content; a rendered voiceover is an `audio` entity.
|
|
1463
|
+
*/
|
|
1464
|
+
const ENTITY_KINDS = Object.freeze({
|
|
1465
|
+
axvideo: Object.freeze([
|
|
1466
|
+
"AXVideo",
|
|
1467
|
+
"Container",
|
|
1468
|
+
"Sequence",
|
|
1469
|
+
"Visual",
|
|
1470
|
+
"Audible"
|
|
1471
|
+
]),
|
|
1472
|
+
timeline: Object.freeze(["Timeline", "Container"]),
|
|
1473
|
+
track: Object.freeze([
|
|
1474
|
+
"Track",
|
|
1475
|
+
"Container",
|
|
1476
|
+
"Sequence"
|
|
1477
|
+
]),
|
|
1478
|
+
clip: Object.freeze(["Clip", "Container"]),
|
|
1479
|
+
asset: Object.freeze(["Asset"]),
|
|
1480
|
+
video: Object.freeze([
|
|
1481
|
+
"Video",
|
|
1482
|
+
"MediaFile",
|
|
1483
|
+
"Sequence",
|
|
1484
|
+
"Visual",
|
|
1485
|
+
"Audible"
|
|
1486
|
+
]),
|
|
1487
|
+
audio: Object.freeze([
|
|
1488
|
+
"Audio",
|
|
1489
|
+
"MediaFile",
|
|
1490
|
+
"Sequence",
|
|
1491
|
+
"Audible"
|
|
1492
|
+
]),
|
|
1493
|
+
image: Object.freeze([
|
|
1494
|
+
"Image",
|
|
1495
|
+
"MediaFile",
|
|
1496
|
+
"Sequence",
|
|
1497
|
+
"Visual"
|
|
1498
|
+
]),
|
|
1499
|
+
voice: Object.freeze([
|
|
1500
|
+
"Voice",
|
|
1501
|
+
"Container",
|
|
1502
|
+
"IPBound"
|
|
1503
|
+
]),
|
|
1504
|
+
"sequence-marker": Object.freeze(["SequenceMarker"]),
|
|
1505
|
+
viewport: Object.freeze(["Viewport"]),
|
|
1506
|
+
"audio-script": Object.freeze(["AudioScript", "SegmentText"]),
|
|
1507
|
+
"phonetic-script": Object.freeze(["PhoneticScript", "SegmentText"]),
|
|
1508
|
+
caption: Object.freeze([
|
|
1509
|
+
"Caption",
|
|
1510
|
+
"SegmentText",
|
|
1511
|
+
"Sequence"
|
|
1512
|
+
])
|
|
1513
|
+
});
|
|
1514
|
+
/** What a known kind declares; extension kinds declare nothing through this table. */
|
|
1515
|
+
function kindsOf(entityKind) {
|
|
1516
|
+
return isKnownEntityKind(entityKind) ? ENTITY_KINDS[entityKind] : [];
|
|
1517
|
+
}
|
|
1518
|
+
function declaresKind(entityKind, kind) {
|
|
1519
|
+
return kindsOf(entityKind).includes(kind);
|
|
1520
|
+
}
|
|
1521
|
+
/**
|
|
1522
|
+
* Sequence admission asks `kinds` first, so an entity that does not declare
|
|
1523
|
+
* `Sequence` can never buy its way in by carrying the fields. Extension kinds
|
|
1524
|
+
* declare nothing and stay structurally detected.
|
|
1525
|
+
*/
|
|
1450
1526
|
function hasSequence(entity) {
|
|
1451
|
-
if (
|
|
1527
|
+
if (isReservedEntityKind(entity.entityKind)) return false;
|
|
1528
|
+
if (isKnownEntityKind(entity.entityKind) && !declaresKind(entity.entityKind, "Sequence")) return false;
|
|
1452
1529
|
return isSequenceFields(entity);
|
|
1453
1530
|
}
|
|
1454
1531
|
function isSequenceFields(value) {
|
|
@@ -1461,22 +1538,16 @@ function isSequenceFields(value) {
|
|
|
1461
1538
|
if (value.sampling !== "native" && value.sampling !== "constant" && value.sampling !== "derived") return false;
|
|
1462
1539
|
return "coordinateSpace" in value && value.coordinateSpace !== void 0;
|
|
1463
1540
|
}
|
|
1464
|
-
function isKnownSequenceKind(kind) {
|
|
1465
|
-
return isMediaAssetVariantKind(kind) || kind === "caption" || kind === "axvideo";
|
|
1466
|
-
}
|
|
1467
1541
|
function isMediaAssetVariantKind(kind) {
|
|
1468
|
-
return kind === "video" || kind === "image" || kind === "audio"
|
|
1542
|
+
return kind === "video" || kind === "image" || kind === "audio";
|
|
1469
1543
|
}
|
|
1470
1544
|
function isKnownEntityKind(kind) {
|
|
1471
|
-
return
|
|
1545
|
+
return Object.hasOwn(ENTITY_KINDS, kind);
|
|
1472
1546
|
}
|
|
1473
1547
|
function isReservedEntityKind(kind) {
|
|
1474
1548
|
const normalized = kind.toLowerCase().replaceAll("-", "").replaceAll("_", "");
|
|
1475
1549
|
return normalized === "speech" || normalized === "videodocument";
|
|
1476
1550
|
}
|
|
1477
|
-
function isKnownNonSequenceKind(kind) {
|
|
1478
|
-
return kind === "timeline" || kind === "track" || kind === "clip" || kind === "asset" || kind === "sequence-marker" || kind === "viewport" || kind === "audio-script" || kind === "phonetic-script";
|
|
1479
|
-
}
|
|
1480
1551
|
function isRecord$1(value) {
|
|
1481
1552
|
return typeof value === "object" && value != null && !Array.isArray(value);
|
|
1482
1553
|
}
|
|
@@ -1660,14 +1731,17 @@ function assembleCaptionContent(rows, captionEntityId) {
|
|
|
1660
1731
|
requireEntityKind(rows, captionEntityId, "caption");
|
|
1661
1732
|
const caption = assembleEntityContent(rows, captionEntityId);
|
|
1662
1733
|
const script = findComposedAudioScript(rows, captionEntityId, "caption");
|
|
1663
|
-
const
|
|
1734
|
+
const selection = captionSelection(caption);
|
|
1735
|
+
const composed = {
|
|
1664
1736
|
...script,
|
|
1665
1737
|
payload: caption.payload
|
|
1666
|
-
}
|
|
1738
|
+
};
|
|
1739
|
+
const segment = selectAudioScriptSegment(composed, selection);
|
|
1667
1740
|
return {
|
|
1668
1741
|
caption,
|
|
1669
1742
|
audioScript: script,
|
|
1670
1743
|
segments: [segment],
|
|
1744
|
+
segmentIndex: scriptSegments(composed).findIndex((entry) => entry.segmentId === selection.segmentId),
|
|
1671
1745
|
text: segment.text
|
|
1672
1746
|
};
|
|
1673
1747
|
}
|
|
@@ -2059,12 +2133,15 @@ function validateKnownEntityPayload(entity) {
|
|
|
2059
2133
|
break;
|
|
2060
2134
|
case "video":
|
|
2061
2135
|
case "audio":
|
|
2062
|
-
|
|
2136
|
+
validateSequencePayload(value, "bounded", "native", problems);
|
|
2137
|
+
validateMediaAssetFields(value, problems);
|
|
2138
|
+
break;
|
|
2063
2139
|
case "caption":
|
|
2064
2140
|
validateSequencePayload(value, "bounded", "native", problems);
|
|
2065
|
-
|
|
2066
|
-
|
|
2067
|
-
|
|
2141
|
+
validateCaptionPayload(value, problems);
|
|
2142
|
+
break;
|
|
2143
|
+
case "voice":
|
|
2144
|
+
validateVoicePayload(value, problems);
|
|
2068
2145
|
break;
|
|
2069
2146
|
case "image":
|
|
2070
2147
|
validateSequencePayload(value, "unbounded", "constant", problems);
|
|
@@ -2171,11 +2248,26 @@ function validateExternalLocator(value, problems) {
|
|
|
2171
2248
|
function validateStorageKey(value, problems) {
|
|
2172
2249
|
if (value.storageKey !== void 0 && (typeof value.storageKey !== "string" || value.storageKey.trim() === "")) problems.push("storageKey must be a non-empty string when present");
|
|
2173
2250
|
}
|
|
2251
|
+
/**
|
|
2252
|
+
* Voice is a timbre identity, not playable content: it owns the voice-library
|
|
2253
|
+
* descriptor and nothing that would make it look like media or a Sequence. The
|
|
2254
|
+
* rendered voiceover is an `audio` entity bound by a `voice-timbre` Relation.
|
|
2255
|
+
*/
|
|
2174
2256
|
function validateVoicePayload(value, problems) {
|
|
2257
|
+
for (const key of [
|
|
2258
|
+
"extent",
|
|
2259
|
+
"sampling",
|
|
2260
|
+
"coordinateSpace",
|
|
2261
|
+
"external",
|
|
2262
|
+
"storageKey"
|
|
2263
|
+
]) if (Object.hasOwn(value, key)) problems.push(`${key} belongs to the rendered Audio, not to the Voice identity`);
|
|
2175
2264
|
const voice = value.voice;
|
|
2176
|
-
if (voice === void 0)
|
|
2265
|
+
if (voice === void 0) {
|
|
2266
|
+
problems.push("voice is required; a Voice entity is its voice-library identity");
|
|
2267
|
+
return;
|
|
2268
|
+
}
|
|
2177
2269
|
if (!isRecord(voice)) {
|
|
2178
|
-
problems.push("voice must be an object
|
|
2270
|
+
problems.push("voice must be an object");
|
|
2179
2271
|
return;
|
|
2180
2272
|
}
|
|
2181
2273
|
if (voice.system !== "voice-library") problems.push("voice.system must be \"voice-library\"");
|
|
@@ -2307,7 +2399,6 @@ function isOwnedLocalEntityValuePath(entityKind, path) {
|
|
|
2307
2399
|
if ([
|
|
2308
2400
|
"video",
|
|
2309
2401
|
"audio",
|
|
2310
|
-
"voice",
|
|
2311
2402
|
"image",
|
|
2312
2403
|
"caption",
|
|
2313
2404
|
"axvideo"
|
|
@@ -2371,7 +2462,7 @@ function collectRelatedEntities(entityRefs, index) {
|
|
|
2371
2462
|
};
|
|
2372
2463
|
}
|
|
2373
2464
|
function expectedSequenceShape(kind) {
|
|
2374
|
-
if (kind === "video" || kind === "audio" || kind === "
|
|
2465
|
+
if (kind === "video" || kind === "audio" || kind === "caption") return {
|
|
2375
2466
|
extent: "bounded",
|
|
2376
2467
|
sampling: "native"
|
|
2377
2468
|
};
|
|
@@ -2412,7 +2503,7 @@ const generatedRelationSpec = Object.freeze({
|
|
|
2412
2503
|
});
|
|
2413
2504
|
const captionAlignmentRelationSpec = Object.freeze({
|
|
2414
2505
|
kind: "caption-alignment",
|
|
2415
|
-
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["caption"]), new Set(["audio"
|
|
2506
|
+
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["caption"]), new Set(["audio"])),
|
|
2416
2507
|
validateMetadata: isCaptionAlignmentMetadata
|
|
2417
2508
|
});
|
|
2418
2509
|
/** `clip-anchor(child, host)` means endpoint 0 follows endpoint 1. */
|
|
@@ -2421,20 +2512,29 @@ const clipAnchorRelationSpec = Object.freeze({
|
|
|
2421
2512
|
validateEndpoints: (endpoints) => endpoints[0].current().entityKind === "clip" && endpoints[1].current().entityKind === "clip",
|
|
2422
2513
|
validateMetadata: isEmptyMetadata
|
|
2423
2514
|
});
|
|
2424
|
-
/**
|
|
2515
|
+
/**
|
|
2516
|
+
* The rendered voiceover Audio and the PhoneticScript it was synthesized from;
|
|
2517
|
+
* kinds determine roles regardless of endpoint positions.
|
|
2518
|
+
*/
|
|
2425
2519
|
const phoneticScriptRenderRelationSpec = Object.freeze({
|
|
2426
2520
|
kind: "phonetic-script-render",
|
|
2427
|
-
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["
|
|
2521
|
+
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio"]), new Set(["phonetic-script"])),
|
|
2522
|
+
validateMetadata: isEmptyMetadata
|
|
2523
|
+
});
|
|
2524
|
+
/**
|
|
2525
|
+
* The timbre identity a rendered voiceover Audio was synthesized with. Voice
|
|
2526
|
+
* stays an identity: it never carries the audio itself, so the link between the
|
|
2527
|
+
* two is a Relation rather than one shared entity.
|
|
2528
|
+
*/
|
|
2529
|
+
const voiceTimbreRelationSpec = Object.freeze({
|
|
2530
|
+
kind: "voice-timbre",
|
|
2531
|
+
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio"]), new Set(["voice"])),
|
|
2428
2532
|
validateMetadata: isEmptyMetadata
|
|
2429
2533
|
});
|
|
2430
2534
|
/** AudioScript was transcribed from Audio, Video, or recorded Voice; kinds determine roles. */
|
|
2431
2535
|
const audioScriptSourceRelationSpec = Object.freeze({
|
|
2432
2536
|
kind: "audio-script-source",
|
|
2433
|
-
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio-script"]), new Set([
|
|
2434
|
-
"audio",
|
|
2435
|
-
"video",
|
|
2436
|
-
"voice"
|
|
2437
|
-
])),
|
|
2537
|
+
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio-script"]), new Set(["audio", "video"])),
|
|
2438
2538
|
validateMetadata: isEmptyMetadata
|
|
2439
2539
|
});
|
|
2440
2540
|
/**
|
|
@@ -2461,6 +2561,7 @@ const builtInRelationSpecs = Object.freeze([
|
|
|
2461
2561
|
captionAlignmentRelationSpec,
|
|
2462
2562
|
clipAnchorRelationSpec,
|
|
2463
2563
|
phoneticScriptRenderRelationSpec,
|
|
2564
|
+
voiceTimbreRelationSpec,
|
|
2464
2565
|
audioScriptSourceRelationSpec,
|
|
2465
2566
|
audioScriptMarkerRelationSpec
|
|
2466
2567
|
]);
|
|
@@ -2490,7 +2591,7 @@ function hasAssetAndSequence(endpoints) {
|
|
|
2490
2591
|
return first.entityKind === "asset" && hasSequence(second) || second.entityKind === "asset" && hasSequence(first);
|
|
2491
2592
|
}
|
|
2492
2593
|
function isGeneratedMedia(entity) {
|
|
2493
|
-
return entity.entityKind === "video" || entity.entityKind === "image" || entity.entityKind === "audio"
|
|
2594
|
+
return entity.entityKind === "video" || entity.entityKind === "image" || entity.entityKind === "audio";
|
|
2494
2595
|
}
|
|
2495
2596
|
function isEmptyMetadata(value) {
|
|
2496
2597
|
return isJsonObject(value) && Object.keys(value).length === 0;
|
|
@@ -2920,7 +3021,7 @@ function resolveGraphLayout(graph) {
|
|
|
2920
3021
|
if (clip.entityKind !== "clip" || graph.adjacent(clip.entityId, "track-clip").length !== 1) fail(`Clip ${clip.entityId} must belong to exactly one Track`);
|
|
2921
3022
|
const marker = graph.one(clip.entityId, "clip-marker", "sequence-marker");
|
|
2922
3023
|
const content = graph.one(marker.entityId, "marker-content");
|
|
2923
|
-
if (!(typedRole === "video_clip" ? ["image", "video"] : typedRole === "speech" ? ["
|
|
3024
|
+
if (!(typedRole === "video_clip" ? ["image", "video"] : typedRole === "speech" ? ["audio"] : typedRole === "caption" ? ["caption"] : ["audio"]).includes(content.entityKind)) fail(`unsupported compatibility content kind ${content.entityKind} on ${typedRole} Track`);
|
|
2924
3025
|
const space = content.payload.coordinateSpace;
|
|
2925
3026
|
if (space !== "ms" && space !== "milliseconds" && (!object(space) || space.unit !== "ms" && space.unit !== "milliseconds")) fail(`media ${content.entityId} must state millisecond coordinates for this reader`);
|
|
2926
3027
|
const source = range(marker.payload.sourceRange, "sourceRange");
|
|
@@ -2999,7 +3100,7 @@ function resolveGraphLayout(graph) {
|
|
|
2999
3100
|
let start = fact.absoluteStart ?? 0;
|
|
3000
3101
|
if (fact.mode === "anchored") {
|
|
3001
3102
|
const host = resolve(fact.anchorId);
|
|
3002
|
-
if (fact.role === "speech" && !["video_clip", "bgm"].includes(facts.get(host.clipEntityId).role)) fail("An anchored
|
|
3103
|
+
if (fact.role === "speech" && !["video_clip", "bgm"].includes(facts.get(host.clipEntityId).role)) fail("An anchored voiceover must use a visual or BGM Clip as its host");
|
|
3003
3104
|
start = host.start + fact.anchorOffset;
|
|
3004
3105
|
} else if (fact.predecessor !== void 0) {
|
|
3005
3106
|
const prior = resolve(fact.predecessor);
|
|
@@ -3083,8 +3184,11 @@ function projectEntityTimeline(rows) {
|
|
|
3083
3184
|
const asset = mediaAssetFacts(graph, content);
|
|
3084
3185
|
requireUntrimmedAudio(content, marker, placement);
|
|
3085
3186
|
const external = externalLocator(asset, "memota-speech");
|
|
3086
|
-
const
|
|
3087
|
-
const
|
|
3187
|
+
const renderEdges = graph.adjacent(content.entityId, "phonetic-script-render");
|
|
3188
|
+
const timbreEdges = graph.adjacent(content.entityId, "voice-timbre");
|
|
3189
|
+
const assembled = renderEdges.length > 0 || timbreEdges.length > 0 ? assemblePhoneticScriptContent(rows, graph.one(content.entityId, "phonetic-script-render", "phonetic-script").entityId) : void 0;
|
|
3190
|
+
if (timbreEdges.length > 1) fail("A voiceover resolves at most one Voice identity");
|
|
3191
|
+
const voice = timbreEdges.length === 0 ? void 0 : graph.one(content.entityId, "voice-timbre", "voice").payload.voice;
|
|
3088
3192
|
if (voice !== void 0 && (!object(voice) || voice.system !== "voice-library" || typeof voice.key !== "string")) fail("Voice must use an explicit voice-library locator");
|
|
3089
3193
|
const captionIds = layout.tracks.flatMap((candidate) => candidate.role === "caption" ? candidate.clips.filter((caption) => caption.anchorClipEntityId === clip.entityId).map((caption) => caption.clipEntityId) : []);
|
|
3090
3194
|
parts[clip.entityId] = { speech: {
|
package/dist/index.d.ts
CHANGED
|
@@ -192,6 +192,16 @@ interface AssembledCaptionContent {
|
|
|
192
192
|
readonly audioScript: EntityRow<'audio-script'>;
|
|
193
193
|
/** Selected AudioScript segments in Caption selection order. */
|
|
194
194
|
readonly segments: readonly ScriptTextSegment[];
|
|
195
|
+
/**
|
|
196
|
+
* Where the selected segment sits in the AudioScript's ordered segments.
|
|
197
|
+
*
|
|
198
|
+
* A Caption locates its text as AudioScript plus index. The selection is
|
|
199
|
+
* stored as the segment's own stable id rather than this ordinal, so a
|
|
200
|
+
* concurrent insert or re-segmentation cannot silently slide a Caption onto
|
|
201
|
+
* different text; the ordinal is derived here for readers that want the
|
|
202
|
+
* position.
|
|
203
|
+
*/
|
|
204
|
+
readonly segmentIndex: number;
|
|
195
205
|
/** Complete display text: the selected segments' text joined in order. */
|
|
196
206
|
readonly text: string;
|
|
197
207
|
}
|
|
@@ -272,17 +282,22 @@ interface AudioMediaAssetFact {
|
|
|
272
282
|
readonly durationMs: number;
|
|
273
283
|
readonly storageKey: string;
|
|
274
284
|
}
|
|
275
|
-
|
|
285
|
+
/**
|
|
286
|
+
* A speech recording. It materializes as an `audio` entity like every other
|
|
287
|
+
* playable sound; `voice` names the timbre identity it was synthesized with,
|
|
288
|
+
* which becomes its own Voice entity linked by a `voice-timbre` Relation.
|
|
289
|
+
*/
|
|
290
|
+
interface VoiceoverMediaAssetFact {
|
|
276
291
|
/** Stable external speech result id, independent of the placed Clip id. */
|
|
277
292
|
readonly assetId: string;
|
|
278
|
-
readonly kind: '
|
|
293
|
+
readonly kind: 'voiceover';
|
|
279
294
|
readonly durationMs: number;
|
|
280
295
|
readonly storageKey: string;
|
|
281
|
-
/** Present for synthesized
|
|
296
|
+
/** Present for synthesized speech, absent for original recorded audio. */
|
|
282
297
|
readonly voice?: VoiceDescriptor;
|
|
283
298
|
}
|
|
284
299
|
/** Facts resolved from media storage. A trim window never substitutes for intrinsic duration. */
|
|
285
|
-
type MediaAssetFact = ImageMediaAssetFact | VideoMediaAssetFact | AudioMediaAssetFact |
|
|
300
|
+
type MediaAssetFact = ImageMediaAssetFact | VideoMediaAssetFact | AudioMediaAssetFact | VoiceoverMediaAssetFact;
|
|
286
301
|
type ClipPlacement = {
|
|
287
302
|
readonly kind: 'sequential';
|
|
288
303
|
readonly order: number;
|
|
@@ -334,7 +349,7 @@ interface DeleteClipTreeInput {
|
|
|
334
349
|
interface VoiceoverCaptionFact {
|
|
335
350
|
/** Stable placed caption identity supplied by the materialized side effect. */
|
|
336
351
|
readonly captionClipEntityId: string;
|
|
337
|
-
/** Directly held bases; includes the AudioScript
|
|
352
|
+
/** Directly held bases; includes the AudioScript the voiceover was rendered from. */
|
|
338
353
|
readonly baseEntityIds: readonly string[];
|
|
339
354
|
/** Ordered selection of AudioScript segments; caption text is never passed inline. */
|
|
340
355
|
readonly selection: CaptionSegmentSelection;
|
|
@@ -345,7 +360,7 @@ interface VoiceoverCaptionFact {
|
|
|
345
360
|
type VoiceoverTakeInput = {
|
|
346
361
|
readonly timelineEntityId: string; /** Stable placed speech identity, distinct from media.assetId. */
|
|
347
362
|
readonly voiceoverClipEntityId: string;
|
|
348
|
-
readonly media:
|
|
363
|
+
readonly media: VoiceoverMediaAssetFact; /** Existing pronunciation variant; its composed AudioScript stays the text owner. */
|
|
349
364
|
readonly phoneticScriptEntityId: string;
|
|
350
365
|
readonly volume: number;
|
|
351
366
|
readonly captions: readonly VoiceoverCaptionFact[];
|
|
@@ -421,8 +436,11 @@ interface TrimClipInput {
|
|
|
421
436
|
}
|
|
422
437
|
interface VoiceoverTakeResult {
|
|
423
438
|
readonly voiceoverClipEntityId: string;
|
|
424
|
-
|
|
425
|
-
|
|
439
|
+
/** The rendered voiceover Audio entity. */
|
|
440
|
+
readonly voiceoverAudioEntityId: string;
|
|
441
|
+
/** The timbre identity it was synthesized with, when the take declared one. */
|
|
442
|
+
readonly voiceEntityId?: string;
|
|
443
|
+
/** The pronunciation variant the voiceover was rendered from. */
|
|
426
444
|
readonly phoneticScriptEntityId: string;
|
|
427
445
|
/** The base-text owner resolved from the PhoneticScript baseEntityIds. */
|
|
428
446
|
readonly audioScriptEntityId: string;
|
|
@@ -556,7 +574,7 @@ declare class EntityTimelineEditor {
|
|
|
556
574
|
insertCaptionClip(input: InsertCaptionClipInput): ClipEntityId;
|
|
557
575
|
private insertPlacedClipIntoDraft;
|
|
558
576
|
private ensureRoleTrack;
|
|
559
|
-
/** The existing pronunciation variant a
|
|
577
|
+
/** The existing pronunciation variant a voiceover consumes; composition is required. */
|
|
560
578
|
private requireComposedPhoneticScript;
|
|
561
579
|
/** Preserve the factual synthesis input, regardless of persisted endpoint positions. */
|
|
562
580
|
private ensurePhoneticRender;
|
|
@@ -632,6 +650,8 @@ interface ImportedMediaAsset {
|
|
|
632
650
|
readonly rows: EntityRelationRows;
|
|
633
651
|
/** The single media variant identity for this asset id; it is its own Asset. */
|
|
634
652
|
readonly contentEntityId: string;
|
|
653
|
+
/** The timbre identity of a synthesized voiceover; absent for recorded media. */
|
|
654
|
+
readonly voiceEntityId?: string;
|
|
635
655
|
}
|
|
636
656
|
/**
|
|
637
657
|
* Import at the editor boundary, not at asset generation time. Existing
|
|
@@ -1716,4 +1736,4 @@ declare function hasAudioScriptAsset(payload: JsonObject): boolean;
|
|
|
1716
1736
|
/** Internal fingerprint prevents a resource from silently describing an older text revision. */
|
|
1717
1737
|
declare function audioScriptAssetFields(payload: JsonObject, assetId: string): JsonObject;
|
|
1718
1738
|
//#endregion
|
|
1719
|
-
export { type Aggregation, type AssembledCaptionContent, type AssembledPhoneticScriptContent, type Attachment, type AudioMediaAssetFact, type BgmPart, type CaptionContentInput, type CaptionDisplayCue, type CaptionPart, type CaptionSegmentSelection, type CaptionStyle, type ClipEntityId, type ClipPlacement, type CommitOptions, DEFAULT_UNIT_TIME_MS, type DeleteBgmInput, type DeleteClipInput, type DeleteClipTreeInput, type DeleteVoiceoverInput, type DerivedItemPosition, type DocStorageLike, type DocVersionMark, type DocumentAudioScript, type DocumentAudioScriptState, type EditorFoundation, EntityDocumentState, type EntityGraphCommitOptions, EntityGraphHttpClient, type EntityGraphState, EntityTimelineEditor, type EntityTimelineIdFactory, EntityTimelineProjectionError, type EntityTimelineTrackRole, type FieldChange, type FieldPath, IMPLEMENTED_SEMANTIC_OP_KINDS, type ImageMediaAssetFact, type ImplementedSemanticOpKind, type ImportedMediaAsset, type InsertCaptionClipInput, type InsertClipInput, type InsertMediaClipInput, type InsertMediaClipsInput, type InsertPlacedClipInput, type JournalEntry, LANE_KINDS_IN_STACK_ORDER, LORO_ENTITY_SCHEMA, type LaneKind, type LinearClipSpeed, LoroEntityDocument, LoroEntityDraft, type MainClipRange, type MakeEmptyPart, ManualSyncDoc, type ManualSyncDocOptions, MedeoHttpDocStorage, type MedeoHttpDocStorageOptions, type MediaAssetFact, type MediaClipInsertion, MemoryDocStorage, MengineAckFailedError, type MengineAuditEntry, type MengineAuditResponse, MengineDocSession, type MengineDocSessionOptions, type MengineDocSessionUpdateEvent, type MengineDocSyncState, type MengineDocumentVersion, type MengineEventStreamOptions, MengineHttpClient, type MengineHttpClientOptions, MengineHttpRequestError, MenginePushRejectedError, type MenginePushResponse, type MenginePushUpdateResponse, type MengineRejectedResponse, type MengineSnapshotResponse, type MengineSseUpdateEvent, type MengineSyncResponse, type MengineUndoState, type MengineUpdateMeta, MirrorVideoDocumentAdapter, type MirrorVideoDocumentOptions, type MoveClipInput, type MoveClipsToStartsInput, type MoveSequentialClipsInput, type MoveVoiceoverInput, type OpActor, type PartAggregation, type PartIdFactory, type PartKind, type PartUnion, type PatchCaptionStyleInput, PlainMemoryAdapter, type PlainMemoryAdapterOptions, type PlannedSemanticOpKind, type PullFailureReason, type PullResult, type PushOutcome, type PushOutcomeKind, type PushResult, type PushResultKind, type ReplaceClipContentInput, type ReplaceMediaClipInput, type ReplaceSequentialClipsInput, type ReplacementMediaClipInput, type ResolvedEntityClipPlacement, type ResolvedEntityTimelineLayout, type ResolvedEntityTrack, SchemaValidator, ScriptCompositionError, type SemanticDocumentAdapter, SemanticEditor, type SemanticOpInput, type SemanticOpKind, type SemanticOpName, type SequentialClipAnchor, type SetBgmInput, type SetCaptionVisibilityInput, type SetClipPlacementInput, type SetClipSpeedInput, type SetClipVolumeInput, type SnapshotReadable, type SolvedVideoDocument, type SpeechHostMap, type SpeechPart, type SpeedShift, TIMELINE_SKELETON_DURATION_MS, type Timeline, type TimelineDoc, type TimelineItem, type Track, type TrackDraft, type TrackItem, type TrackItemDraft, type TrackItemTimePosition, type TransactAudit, type TrimClipInput, type UpdateClipInput, type UpdateClipMarkerInput, ValidationError, type VideoClipPart, type VideoDocument, type VideoDocumentDraft, type VideoDocumentMirrorSchema, VideoDocumentValidationError, type VideoDocumentValidationIssue, type VideoDocumentValidationIssueCode, type VideoDraft, type CaptionPart$1 as VideoDraftCaptionPart, type VideoDraftContent, type VideoDraftPartUnion, type Timeline$1 as VideoDraftTimeline, type Track$1 as VideoDraftTrack, type TrackItem$1 as VideoDraftTrackItem, type VideoMediaAssetFact, type VisualMediaAssetFact, type
|
|
1739
|
+
export { type Aggregation, type AssembledCaptionContent, type AssembledPhoneticScriptContent, type Attachment, type AudioMediaAssetFact, type BgmPart, type CaptionContentInput, type CaptionDisplayCue, type CaptionPart, type CaptionSegmentSelection, type CaptionStyle, type ClipEntityId, type ClipPlacement, type CommitOptions, DEFAULT_UNIT_TIME_MS, type DeleteBgmInput, type DeleteClipInput, type DeleteClipTreeInput, type DeleteVoiceoverInput, type DerivedItemPosition, type DocStorageLike, type DocVersionMark, type DocumentAudioScript, type DocumentAudioScriptState, type EditorFoundation, EntityDocumentState, type EntityGraphCommitOptions, EntityGraphHttpClient, type EntityGraphState, EntityTimelineEditor, type EntityTimelineIdFactory, EntityTimelineProjectionError, type EntityTimelineTrackRole, type FieldChange, type FieldPath, IMPLEMENTED_SEMANTIC_OP_KINDS, type ImageMediaAssetFact, type ImplementedSemanticOpKind, type ImportedMediaAsset, type InsertCaptionClipInput, type InsertClipInput, type InsertMediaClipInput, type InsertMediaClipsInput, type InsertPlacedClipInput, type JournalEntry, LANE_KINDS_IN_STACK_ORDER, LORO_ENTITY_SCHEMA, type LaneKind, type LinearClipSpeed, LoroEntityDocument, LoroEntityDraft, type MainClipRange, type MakeEmptyPart, ManualSyncDoc, type ManualSyncDocOptions, MedeoHttpDocStorage, type MedeoHttpDocStorageOptions, type MediaAssetFact, type MediaClipInsertion, MemoryDocStorage, MengineAckFailedError, type MengineAuditEntry, type MengineAuditResponse, MengineDocSession, type MengineDocSessionOptions, type MengineDocSessionUpdateEvent, type MengineDocSyncState, type MengineDocumentVersion, type MengineEventStreamOptions, MengineHttpClient, type MengineHttpClientOptions, MengineHttpRequestError, MenginePushRejectedError, type MenginePushResponse, type MenginePushUpdateResponse, type MengineRejectedResponse, type MengineSnapshotResponse, type MengineSseUpdateEvent, type MengineSyncResponse, type MengineUndoState, type MengineUpdateMeta, MirrorVideoDocumentAdapter, type MirrorVideoDocumentOptions, type MoveClipInput, type MoveClipsToStartsInput, type MoveSequentialClipsInput, type MoveVoiceoverInput, type OpActor, type PartAggregation, type PartIdFactory, type PartKind, type PartUnion, type PatchCaptionStyleInput, PlainMemoryAdapter, type PlainMemoryAdapterOptions, type PlannedSemanticOpKind, type PullFailureReason, type PullResult, type PushOutcome, type PushOutcomeKind, type PushResult, type PushResultKind, type ReplaceClipContentInput, type ReplaceMediaClipInput, type ReplaceSequentialClipsInput, type ReplacementMediaClipInput, type ResolvedEntityClipPlacement, type ResolvedEntityTimelineLayout, type ResolvedEntityTrack, SchemaValidator, ScriptCompositionError, type SemanticDocumentAdapter, SemanticEditor, type SemanticOpInput, type SemanticOpKind, type SemanticOpName, type SequentialClipAnchor, type SetBgmInput, type SetCaptionVisibilityInput, type SetClipPlacementInput, type SetClipSpeedInput, type SetClipVolumeInput, type SnapshotReadable, type SolvedVideoDocument, type SpeechHostMap, type SpeechPart, type SpeedShift, TIMELINE_SKELETON_DURATION_MS, type Timeline, type TimelineDoc, type TimelineItem, type Track, type TrackDraft, type TrackItem, type TrackItemDraft, type TrackItemTimePosition, type TransactAudit, type TrimClipInput, type UpdateClipInput, type UpdateClipMarkerInput, ValidationError, type VideoClipPart, type VideoDocument, type VideoDocumentDraft, type VideoDocumentMirrorSchema, VideoDocumentValidationError, type VideoDocumentValidationIssue, type VideoDocumentValidationIssueCode, type VideoDraft, type CaptionPart$1 as VideoDraftCaptionPart, type VideoDraftContent, type VideoDraftPartUnion, type Timeline$1 as VideoDraftTimeline, type Track$1 as VideoDraftTrack, type TrackItem$1 as VideoDraftTrackItem, type VideoMediaAssetFact, type VisualMediaAssetFact, type VoiceoverCaptionFact, type VoiceoverMediaAssetFact, type VoiceoverTakeInput, type VoiceoverTakeResult, type WaitForServerAckOptions, applyEntityRows, applyFieldChanges, arrangeMainTrackSeamlessly, assembleCaptionContent, assembleEntityContent, assemblePhoneticScriptContent, assembleScriptText, assertCanonicalEditorResources, assertEntityProjectionPreservesLegacy, assertFieldChanges, assertMediaAssetWritePolicy, assertValidVideoDocument, audioScriptAssetContent, audioScriptAssetFields, audioScriptAssetHash, base64ToBytes, buildInitialVideoDocument, buildSpeechHostMap, bytesToBase64, cascadeAfterVideoClipChanges, compileEntityRows, createEditSandbox, createMirrorVideoDocument, createMirrorVideoDocumentAdapter, createPlainMemoryAdapter, decodeDocVersionMark, derivePositionFromAbs, effectiveVideoClipDurationMs, encodeDocVersionMark, ensureEditorFoundation, ensureLaneTrack, fillMainTrackTimeGaps, findComposedAudioScript, findLaneTrack, fromVideoDocument, generatePartId, getAt, hasAudioScriptAsset, hostForAbsMs, importMediaAsset, isEmptyVideoClip, isImplementedSemanticOpKind, isMap, laneTrackId, mainTrackRanges, migrateLegacyTimelineToEntities, partDurationMs, partUnionSchema, prepareCaptionContent, projectEntityTimeline, readDocumentAudioScript, readEntityDocument, readMainTrackItems, readMengineEventStream, readPart, readPartDurationMs, readVideoDocumentFromDraft, reassignSpeechesToVideoClipsByTime, recalculateTimelineDuration, relativePositionForAbs, replayJournal, resolveAllSpeechOverlaps, resolveEntityTimelineLayout, resolveFieldPath, resolveSpeechOverlapByShiftingVideos, safeDurationMs, index_d_exports as schemas, selectAudioScriptSegment, snapshotToPlain, solveVideoDocument, speedOf, syncAggregatedClipsTimePosition, toVideoDocument, validateVideoDocument, videoDocumentMirrorSchema, videoDocumentSchema };
|
package/dist/index.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { $ as findLaneTrack, A as assemblePhoneticScriptContent, B as isMediaAssetVariantKind, C as EntityProjectionGraph, D as ScriptCompositionError, E as decodeEntityRelationRows, F as variantBaseEntityIds, G as toVideoDocument, H as buildSpeechHostMap, I as isJsonObject, J as validateVideoDocument, K as VideoDocumentValidationError, L as createEntityId, M as findComposedAudioScript, N as selectAudioScriptSegment, O as assembleCaptionContent, P as updateEntityFields, Q as ensureLaneTrack, R as createRelationId, S as projectEntityTimeline, St as bytesToBase64, T as EntityTimelineProjectionError, U as derivePositionFromAbs, V as buildInitialVideoDocument, W as fromVideoDocument, X as videoDocumentSchema, Y as partUnionSchema, Z as LANE_KINDS_IN_STACK_ORDER, _ as hasAudioScriptAsset, _t as speedOf, a as createMirrorVideoDocument, at as fillMainTrackTimeGaps, b as assertFieldChanges, bt as videoDocumentMirrorSchema, c as DocumentMutationGuard, ct as resolveAllSpeechOverlaps, d as LoroEntityDraft, dt as TIMELINE_SKELETON_DURATION_MS, et as hasResolvedPlacement, f as readEntityDocument, ft as isEmptyVideoClip, g as audioScriptAssetHash, gt as effectiveVideoClipDurationMs, h as audioScriptAssetFields, ht as DEFAULT_UNIT_TIME_MS, i as MirrorVideoDocumentAdapter, it as arrangeMainTrackSeamlessly, j as assembleScriptText, k as assembleEntityContent, l as LORO_ENTITY_SCHEMA, lt as resolveSpeechOverlapByShiftingVideos, m as audioScriptAssetContent, mt as safeDurationMs, n as createPlainMemoryAdapter, nt as solveVideoDocument, o as createMirrorVideoDocumentAdapter, ot as reassignSpeechesToVideoClipsByTime, p as entityOrderGroups, pt as partDurationMs, q as assertValidVideoDocument, r as generatePartId, rt as cascadeAfterVideoClipChanges, s as readVideoDocumentFromDraft, st as recalculateTimelineDuration, t as PlainMemoryAdapter, tt as laneTrackId, u as LoroEntityDocument, ut as syncAggregatedClipsTimePosition, v as equalJson, vt as partUnionToDraft, w as resolveEntityTimelineLayout, x as resolveFieldPath, xt as base64ToBytes, y as applyFieldChanges, yt as recordEntries, z as hasSequence } from "./document-
|
|
1
|
+
import { $ as findLaneTrack, A as assemblePhoneticScriptContent, B as isMediaAssetVariantKind, C as EntityProjectionGraph, D as ScriptCompositionError, E as decodeEntityRelationRows, F as variantBaseEntityIds, G as toVideoDocument, H as buildSpeechHostMap, I as isJsonObject, J as validateVideoDocument, K as VideoDocumentValidationError, L as createEntityId, M as findComposedAudioScript, N as selectAudioScriptSegment, O as assembleCaptionContent, P as updateEntityFields, Q as ensureLaneTrack, R as createRelationId, S as projectEntityTimeline, St as bytesToBase64, T as EntityTimelineProjectionError, U as derivePositionFromAbs, V as buildInitialVideoDocument, W as fromVideoDocument, X as videoDocumentSchema, Y as partUnionSchema, Z as LANE_KINDS_IN_STACK_ORDER, _ as hasAudioScriptAsset, _t as speedOf, a as createMirrorVideoDocument, at as fillMainTrackTimeGaps, b as assertFieldChanges, bt as videoDocumentMirrorSchema, c as DocumentMutationGuard, ct as resolveAllSpeechOverlaps, d as LoroEntityDraft, dt as TIMELINE_SKELETON_DURATION_MS, et as hasResolvedPlacement, f as readEntityDocument, ft as isEmptyVideoClip, g as audioScriptAssetHash, gt as effectiveVideoClipDurationMs, h as audioScriptAssetFields, ht as DEFAULT_UNIT_TIME_MS, i as MirrorVideoDocumentAdapter, it as arrangeMainTrackSeamlessly, j as assembleScriptText, k as assembleEntityContent, l as LORO_ENTITY_SCHEMA, lt as resolveSpeechOverlapByShiftingVideos, m as audioScriptAssetContent, mt as safeDurationMs, n as createPlainMemoryAdapter, nt as solveVideoDocument, o as createMirrorVideoDocumentAdapter, ot as reassignSpeechesToVideoClipsByTime, p as entityOrderGroups, pt as partDurationMs, q as assertValidVideoDocument, r as generatePartId, rt as cascadeAfterVideoClipChanges, s as readVideoDocumentFromDraft, st as recalculateTimelineDuration, t as PlainMemoryAdapter, tt as laneTrackId, u as LoroEntityDocument, ut as syncAggregatedClipsTimePosition, v as equalJson, vt as partUnionToDraft, w as resolveEntityTimelineLayout, x as resolveFieldPath, xt as base64ToBytes, y as applyFieldChanges, yt as recordEntries, z as hasSequence } from "./document-C30rnewk.js";
|
|
2
2
|
import { addSpeechesInputSchema, addVideoClipsInputSchema, adjustBgmVolumeInputSchema, adjustSpeechVolumeInputSchema, adjustVideoClipDurationInputSchema, adjustVideoClipVolumeInputSchema, changeSpeechScriptInputSchema, changeSpeechVoiceInputSchema, deleteBgmInputSchema, deleteSpeechesInputSchema, deleteVideoClipsInputSchema, moveSpeechesInputSchema, moveVideoClipsByAnchorInputSchema, moveVideoClipsInputSchema, replaceVideoClipContentInputSchema, replaceVideoClipSequenceInputSchema, setBgmInputSchema, setCaptionStyleInputSchema, setCaptionVisibilityInputSchema, setVideoClipSpeedShiftInputSchema, t as schemas_exports } from "./schemas.js";
|
|
3
3
|
import { LoroDoc, UndoManager, VersionVector } from "loro-crdt";
|
|
4
4
|
import { ClientServerSynchronizer, DocManager } from "@mengine/sync";
|
|
@@ -579,36 +579,55 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
|
|
|
579
579
|
"image",
|
|
580
580
|
"video",
|
|
581
581
|
"audio",
|
|
582
|
-
"
|
|
582
|
+
"voiceover"
|
|
583
583
|
].includes(fact.kind)) throw new Error("Unsupported media kind");
|
|
584
584
|
if (fact.kind !== "image" && (!Number.isFinite(fact.durationMs) || fact.durationMs <= 0)) throw new Error(`${fact.kind} ${fact.assetId} requires its factual positive duration`);
|
|
585
|
-
if ((fact.kind === "audio" || fact.kind === "
|
|
585
|
+
if ((fact.kind === "audio" || fact.kind === "voiceover") && fact.storageKey.trim() === "") throw new Error(`${fact.kind} ${fact.assetId} requires its physical storage key`);
|
|
586
586
|
decodeEntityRelationRows(initial, numericCoordinates);
|
|
587
587
|
assertCanonicalEditorResources(initial);
|
|
588
588
|
const entities = structuredClone([...initial.entities]);
|
|
589
589
|
const relations = structuredClone([...initial.relations]);
|
|
590
|
-
const system = fact.kind === "
|
|
590
|
+
const system = fact.kind === "voiceover" ? "memota-speech" : "memota";
|
|
591
|
+
const entityKind = fact.kind === "voiceover" ? "audio" : fact.kind;
|
|
592
|
+
const timbre = fact.kind === "voiceover" ? fact.voice : void 0;
|
|
591
593
|
const existing = entities.filter((row) => {
|
|
592
594
|
const external = row.payload.external;
|
|
593
595
|
return isMediaAssetVariantKind(row.entityKind) && isJsonObject(external) && external.system === system && external.key === fact.assetId;
|
|
594
596
|
});
|
|
597
|
+
const requireMatchingVariant = (row) => {
|
|
598
|
+
const extent = row.payload.extent;
|
|
599
|
+
if (row.entityKind !== entityKind || fact.kind !== "image" && (!isJsonObject(extent) || extent.start !== 0 || extent.end !== fact.durationMs)) throw new Error(`Media facts conflict with existing identity for ${fact.assetId}`);
|
|
600
|
+
if (fact.storageKey !== void 0 && row.payload.storageKey !== fact.storageKey) {
|
|
601
|
+
if (row.payload.storageKey !== void 0) throw new Error(`Media facts conflict with existing storage key for ${fact.assetId}`);
|
|
602
|
+
return {
|
|
603
|
+
...row,
|
|
604
|
+
payload: {
|
|
605
|
+
...row.payload,
|
|
606
|
+
storageKey: fact.storageKey
|
|
607
|
+
}
|
|
608
|
+
};
|
|
609
|
+
}
|
|
610
|
+
return row;
|
|
611
|
+
};
|
|
595
612
|
if (existing.length > 1) throw new Error(`Duplicate media identities for ${fact.assetId}; reuse ${existing[0].entityId}`);
|
|
596
613
|
if (existing[0] !== void 0) {
|
|
597
|
-
const variant = requireMatchingVariant(existing[0]
|
|
614
|
+
const variant = requireMatchingVariant(existing[0]);
|
|
598
615
|
replaceEntity$1(entities, variant.entityId, variant);
|
|
616
|
+
const voiceEntityId = bindVoiceTimbre(entities, relations, variant.entityId, timbre, idFactory);
|
|
599
617
|
return {
|
|
600
618
|
rows: {
|
|
601
619
|
entities,
|
|
602
620
|
relations
|
|
603
621
|
},
|
|
604
|
-
contentEntityId: variant.entityId
|
|
622
|
+
contentEntityId: variant.entityId,
|
|
623
|
+
...voiceEntityId
|
|
605
624
|
};
|
|
606
625
|
}
|
|
607
626
|
const entityId = createEntityId(idFactory("entity"));
|
|
608
627
|
if (entities.some((row) => row.entityId === entityId)) throw new Error(`Duplicate entity id ${entityId}`);
|
|
609
628
|
const variant = {
|
|
610
629
|
entityId,
|
|
611
|
-
entityKind
|
|
630
|
+
entityKind,
|
|
612
631
|
payload: {
|
|
613
632
|
extent: fact.kind === "image" ? {
|
|
614
633
|
kind: "unbounded",
|
|
@@ -624,15 +643,11 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
|
|
|
624
643
|
system,
|
|
625
644
|
key: fact.assetId
|
|
626
645
|
},
|
|
627
|
-
...fact.storageKey === void 0 ? {} : { storageKey: fact.storageKey }
|
|
628
|
-
...fact.kind === "voice" && fact.voice !== void 0 ? { voice: {
|
|
629
|
-
system: fact.voice.system,
|
|
630
|
-
key: fact.voice.key,
|
|
631
|
-
...fact.voice.name === void 0 ? {} : { name: fact.voice.name }
|
|
632
|
-
} } : {}
|
|
646
|
+
...fact.storageKey === void 0 ? {} : { storageKey: fact.storageKey }
|
|
633
647
|
}
|
|
634
648
|
};
|
|
635
649
|
entities.push(variant);
|
|
650
|
+
const voiceEntityId = bindVoiceTimbre(entities, relations, entityId, timbre, idFactory);
|
|
636
651
|
const rows = {
|
|
637
652
|
entities,
|
|
638
653
|
relations
|
|
@@ -641,33 +656,58 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
|
|
|
641
656
|
assertCanonicalEditorResources(rows);
|
|
642
657
|
return {
|
|
643
658
|
rows,
|
|
644
|
-
contentEntityId: entityId
|
|
659
|
+
contentEntityId: entityId,
|
|
660
|
+
...voiceEntityId
|
|
645
661
|
};
|
|
646
662
|
}
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
};
|
|
663
|
+
/**
|
|
664
|
+
* Attach the timbre a synthesized voiceover was rendered with. One Voice
|
|
665
|
+
* identity is shared by every take naming the same library entry, so an
|
|
666
|
+
* existing row is reused rather than duplicated. Recorded media declares no
|
|
667
|
+
* timbre and keeps no link.
|
|
668
|
+
*/
|
|
669
|
+
function bindVoiceTimbre(entities, relations, audioEntityId, descriptor, idFactory) {
|
|
670
|
+
const linked = relations.filter((relation) => relation.relationKind === "voice-timbre" && (relation.endpoint0EntityId === audioEntityId || relation.endpoint1EntityId === audioEntityId));
|
|
671
|
+
if (linked.length > 1) throw new Error(`Voiceover ${audioEntityId} has multiple Voice timbre relations`);
|
|
672
|
+
if (descriptor === void 0) {
|
|
673
|
+
if (linked.length > 0) throw new Error(`Recorded media ${audioEntityId} cannot keep a synthesized Voice timbre`);
|
|
674
|
+
return {};
|
|
659
675
|
}
|
|
660
|
-
|
|
676
|
+
const voice = entities.find((row) => row.entityKind === "voice" && sameVoiceDescriptor(row.payload.voice, descriptor)) ?? appendVoiceIdentity(entities, descriptor, idFactory);
|
|
677
|
+
const existing = linked[0];
|
|
678
|
+
if (existing === void 0) relations.push({
|
|
679
|
+
relationId: createRelationId(idFactory("relation")),
|
|
680
|
+
relationKind: "voice-timbre",
|
|
681
|
+
endpoint0EntityId: createEntityId(audioEntityId),
|
|
682
|
+
endpoint1EntityId: createEntityId(voice.entityId),
|
|
683
|
+
metadata: {},
|
|
684
|
+
trace: {}
|
|
685
|
+
});
|
|
686
|
+
else if (existing.endpoint0EntityId !== audioEntityId || existing.endpoint1EntityId !== voice.entityId) throw new Error(`Voiceover ${audioEntityId} already carries a different Voice timbre`);
|
|
687
|
+
return { voiceEntityId: voice.entityId };
|
|
688
|
+
}
|
|
689
|
+
function appendVoiceIdentity(entities, descriptor, idFactory) {
|
|
690
|
+
const voice = {
|
|
691
|
+
entityId: createEntityId(idFactory("entity")),
|
|
692
|
+
entityKind: "voice",
|
|
693
|
+
payload: { voice: {
|
|
694
|
+
system: descriptor.system,
|
|
695
|
+
key: descriptor.key,
|
|
696
|
+
...descriptor.name === void 0 ? {} : { name: descriptor.name }
|
|
697
|
+
} }
|
|
698
|
+
};
|
|
699
|
+
entities.push(voice);
|
|
700
|
+
return voice;
|
|
701
|
+
}
|
|
702
|
+
/** Two takes share a Voice identity only when they name the same library entry. */
|
|
703
|
+
function sameVoiceDescriptor(stored, descriptor) {
|
|
704
|
+
return isJsonObject(stored) && stored.system === descriptor.system && stored.key === descriptor.key && stored.name === descriptor.name;
|
|
661
705
|
}
|
|
662
706
|
function replaceEntity$1(entities, entityId, replacement) {
|
|
663
707
|
const index = entities.findIndex((row) => row.entityId === entityId);
|
|
664
708
|
if (index < 0) throw new Error(`Entity id ${entityId} does not exist`);
|
|
665
709
|
entities[index] = replacement;
|
|
666
710
|
}
|
|
667
|
-
function sameVoice(left, right) {
|
|
668
|
-
if (right === void 0) return left === void 0;
|
|
669
|
-
return isJsonObject(left) && left.system === right.system && left.key === right.key && left.name === right.name && Object.keys(left).every((key) => key === "system" || key === "key" || key === "name");
|
|
670
|
-
}
|
|
671
711
|
//#endregion
|
|
672
712
|
//#region src/entity-editor/entity-timeline-editor.ts
|
|
673
713
|
/**
|
|
@@ -1203,20 +1243,20 @@ var EntityTimelineEditor = class {
|
|
|
1203
1243
|
hostClipEntityId: input.hostClipEntityId,
|
|
1204
1244
|
anchorOffset: input.anchorOffset
|
|
1205
1245
|
};
|
|
1206
|
-
if (input.placement !== void 0 && (input.hostClipEntityId !== void 0 || input.anchorOffset !== void 0)) throw new Error("Specify
|
|
1246
|
+
if (input.placement !== void 0 && (input.hostClipEntityId !== void 0 || input.anchorOffset !== void 0)) throw new Error("Specify voiceover placement or hostClipEntityId/anchorOffset, not both");
|
|
1207
1247
|
requireVolume(input.volume);
|
|
1208
1248
|
const imported = importMediaAsset(draft, input.media, this.idFactory);
|
|
1209
1249
|
replaceRows(draft, imported.rows);
|
|
1210
|
-
const
|
|
1250
|
+
const voiceover = requireEntityKind(draft, imported.contentEntityId, "audio");
|
|
1211
1251
|
const phonetic = this.requireComposedPhoneticScript(draft, input.phoneticScriptEntityId);
|
|
1212
|
-
this.ensurePhoneticRender(draft,
|
|
1252
|
+
this.ensurePhoneticRender(draft, voiceover.entityId, phonetic.entityId);
|
|
1213
1253
|
const speechTrack = this.ensureRoleTrack(draft, input.timelineEntityId, "speech");
|
|
1214
1254
|
const existing = draft.entities.find((row) => row.entityId === input.voiceoverClipEntityId);
|
|
1215
1255
|
let voiceoverClipEntityId;
|
|
1216
1256
|
if (existing === void 0) voiceoverClipEntityId = this.insertPlacedClipIntoDraft(draft, {
|
|
1217
1257
|
clipEntityId: input.voiceoverClipEntityId,
|
|
1218
1258
|
trackEntityId: speechTrack.entityId,
|
|
1219
|
-
contentEntityId:
|
|
1259
|
+
contentEntityId: voiceover.entityId,
|
|
1220
1260
|
sourceRange: {
|
|
1221
1261
|
start: 0,
|
|
1222
1262
|
end: input.media.durationMs
|
|
@@ -1240,7 +1280,7 @@ var EntityTimelineEditor = class {
|
|
|
1240
1280
|
});
|
|
1241
1281
|
this.relinkRelation(draft, topology.markerContentRelation, {
|
|
1242
1282
|
endpoint0EntityId: topology.marker.entityId,
|
|
1243
|
-
endpoint1EntityId:
|
|
1283
|
+
endpoint1EntityId: voiceover.entityId
|
|
1244
1284
|
});
|
|
1245
1285
|
this.applyClipPlacement(draft, topology.clip.entityId, placement);
|
|
1246
1286
|
const placed = requireClipTopology(draft, topology.clip.entityId);
|
|
@@ -1259,11 +1299,12 @@ var EntityTimelineEditor = class {
|
|
|
1259
1299
|
}
|
|
1260
1300
|
});
|
|
1261
1301
|
}
|
|
1262
|
-
this.reconcileVoiceoverCaptions(draft, voiceoverClipEntityId,
|
|
1302
|
+
this.reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voiceover.entityId, phonetic.entityId, input.captions);
|
|
1263
1303
|
this.resolveOneVoiceoverOverlap(draft);
|
|
1264
1304
|
return {
|
|
1265
1305
|
voiceoverClipEntityId,
|
|
1266
|
-
|
|
1306
|
+
voiceoverAudioEntityId: voiceover.entityId,
|
|
1307
|
+
...imported.voiceEntityId === void 0 ? {} : { voiceEntityId: imported.voiceEntityId },
|
|
1267
1308
|
phoneticScriptEntityId: phonetic.entityId,
|
|
1268
1309
|
audioScriptEntityId: requireComposedAudioScript(draft, phonetic.entityId),
|
|
1269
1310
|
captionClipEntityIds: input.captions.map((caption) => caption.captionClipEntityId)
|
|
@@ -1469,23 +1510,23 @@ var EntityTimelineEditor = class {
|
|
|
1469
1510
|
draft.relations.push(this.createRelation(draft, "timeline-track", timeline.entityId, track.entityId));
|
|
1470
1511
|
return track;
|
|
1471
1512
|
}
|
|
1472
|
-
/** The existing pronunciation variant a
|
|
1513
|
+
/** The existing pronunciation variant a voiceover consumes; composition is required. */
|
|
1473
1514
|
requireComposedPhoneticScript(draft, phoneticScriptEntityId) {
|
|
1474
1515
|
const phonetic = requireEntityKind(draft, phoneticScriptEntityId, "phonetic-script");
|
|
1475
1516
|
requireComposedAudioScript(draft, phonetic.entityId);
|
|
1476
1517
|
return phonetic;
|
|
1477
1518
|
}
|
|
1478
1519
|
/** Preserve the factual synthesis input, regardless of persisted endpoint positions. */
|
|
1479
|
-
ensurePhoneticRender(draft,
|
|
1480
|
-
const renders = draft.relations.filter((relation) => relation.relationKind === "phonetic-script-render" && (relation.endpoint0EntityId ===
|
|
1481
|
-
if (renders.length > 1) throw new Error(`
|
|
1520
|
+
ensurePhoneticRender(draft, voiceoverAudioEntityId, phoneticScriptEntityId) {
|
|
1521
|
+
const renders = draft.relations.filter((relation) => relation.relationKind === "phonetic-script-render" && (relation.endpoint0EntityId === voiceoverAudioEntityId || relation.endpoint1EntityId === voiceoverAudioEntityId));
|
|
1522
|
+
if (renders.length > 1) throw new Error(`Voiceover "${voiceoverAudioEntityId}" has multiple PhoneticScript render relations`);
|
|
1482
1523
|
if (renders[0] !== void 0) {
|
|
1483
|
-
if (otherEndpoint(renders[0],
|
|
1484
|
-
throw new Error(`
|
|
1524
|
+
if (otherEndpoint(renders[0], voiceoverAudioEntityId) === phoneticScriptEntityId) return;
|
|
1525
|
+
throw new Error(`Voiceover "${voiceoverAudioEntityId}" already has a different PhoneticScript synthesis input`);
|
|
1485
1526
|
}
|
|
1486
|
-
draft.relations.push(this.createRelation(draft, "phonetic-script-render",
|
|
1527
|
+
draft.relations.push(this.createRelation(draft, "phonetic-script-render", voiceoverAudioEntityId, phoneticScriptEntityId));
|
|
1487
1528
|
}
|
|
1488
|
-
reconcileVoiceoverCaptions(draft, voiceoverClipEntityId,
|
|
1529
|
+
reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voiceoverAudioEntityId, phoneticScriptEntityId, captions) {
|
|
1489
1530
|
const audioScriptEntityId = requireComposedAudioScript(draft, phoneticScriptEntityId);
|
|
1490
1531
|
const requested = new Set(captions.map((caption) => caption.captionClipEntityId));
|
|
1491
1532
|
if (requested.size !== captions.length) throw new Error("Caption Clip ids must be unique within a voiceover take");
|
|
@@ -1516,7 +1557,7 @@ var EntityTimelineEditor = class {
|
|
|
1516
1557
|
};
|
|
1517
1558
|
draft.entities.push(content);
|
|
1518
1559
|
assertCaptionScript(draft, content.entityId, audioScriptEntityId);
|
|
1519
|
-
draft.relations.push(this.createRelation(draft, "caption-alignment", content.entityId,
|
|
1560
|
+
draft.relations.push(this.createRelation(draft, "caption-alignment", content.entityId, voiceoverAudioEntityId, { alignment: {
|
|
1520
1561
|
start: startMs,
|
|
1521
1562
|
end: startMs + durationMs
|
|
1522
1563
|
} }));
|
|
@@ -1573,7 +1614,7 @@ var EntityTimelineEditor = class {
|
|
|
1573
1614
|
}
|
|
1574
1615
|
}
|
|
1575
1616
|
});
|
|
1576
|
-
this.replaceCaptionAlignment(draft, content.entityId,
|
|
1617
|
+
this.replaceCaptionAlignment(draft, content.entityId, voiceoverAudioEntityId, startMs, durationMs);
|
|
1577
1618
|
}
|
|
1578
1619
|
}
|
|
1579
1620
|
ensureRoleTrackForClip(draft, clipEntityId, role) {
|
|
@@ -1582,10 +1623,10 @@ var EntityTimelineEditor = class {
|
|
|
1582
1623
|
const timeline = requireEntityKind(draft, otherEndpoint(requireOneRelation(draft, track.entityId, "timeline-track"), track.entityId), "timeline");
|
|
1583
1624
|
return this.ensureRoleTrack(draft, timeline.entityId, role);
|
|
1584
1625
|
}
|
|
1585
|
-
replaceCaptionAlignment(draft, captionEntityId,
|
|
1626
|
+
replaceCaptionAlignment(draft, captionEntityId, voiceoverAudioEntityId, startMs, durationMs) {
|
|
1586
1627
|
const removed = new Set(incidentRelations(draft, captionEntityId).filter((relation) => relation.relationKind === "caption-alignment").map((relation) => relation.relationId));
|
|
1587
1628
|
draft.relations = draft.relations.filter((relation) => !removed.has(relation.relationId));
|
|
1588
|
-
draft.relations.push(this.createRelation(draft, "caption-alignment", captionEntityId,
|
|
1629
|
+
draft.relations.push(this.createRelation(draft, "caption-alignment", captionEntityId, voiceoverAudioEntityId, { alignment: {
|
|
1589
1630
|
start: startMs,
|
|
1590
1631
|
end: startMs + durationMs
|
|
1591
1632
|
} }));
|
|
@@ -2048,7 +2089,7 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
|
|
|
2048
2089
|
relations
|
|
2049
2090
|
}, {
|
|
2050
2091
|
assetId: speech.origin_speech_id,
|
|
2051
|
-
kind: "
|
|
2092
|
+
kind: "voiceover",
|
|
2052
2093
|
durationMs: speech.media_duration_ms,
|
|
2053
2094
|
storageKey: speech.audio_storage_key,
|
|
2054
2095
|
voice: {
|
|
@@ -2136,8 +2177,8 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
|
|
|
2136
2177
|
} else if (part.caption !== void 0) {
|
|
2137
2178
|
const caption = part.caption;
|
|
2138
2179
|
duration = caption.initial_duration_ms;
|
|
2139
|
-
const
|
|
2140
|
-
const render = relations.find((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId ===
|
|
2180
|
+
const voiceover = contentByPartId.get(caption.speech_part_id);
|
|
2181
|
+
const render = relations.find((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voiceover);
|
|
2141
2182
|
const phonetic = entities.find((entity) => entity.entityId === render?.endpoint1EntityId);
|
|
2142
2183
|
const script = phonetic === void 0 ? void 0 : composedAudioScript(phonetic, entities);
|
|
2143
2184
|
if (script === void 0) fail("Legacy Caption has no composed AudioScript");
|
|
@@ -2164,7 +2205,7 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
|
|
|
2164
2205
|
selection,
|
|
2165
2206
|
...caption.style === void 0 ? {} : { style: toEntityCaptionStyle(caption.style) }
|
|
2166
2207
|
});
|
|
2167
|
-
link("caption-alignment", content,
|
|
2208
|
+
link("caption-alignment", content, voiceover, { alignment: {
|
|
2168
2209
|
start: caption.start_ms,
|
|
2169
2210
|
end: caption.start_ms + duration
|
|
2170
2211
|
} });
|
|
@@ -2503,21 +2544,21 @@ function requirePart(document, partId) {
|
|
|
2503
2544
|
if (part === void 0) fail(`Placed Clip ${partId} has no part payload`);
|
|
2504
2545
|
return part;
|
|
2505
2546
|
}
|
|
2506
|
-
function ensureSpeechPhoneticChain(
|
|
2507
|
-
const renders = relations.filter((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId ===
|
|
2508
|
-
if (renders.length > 1) fail(`
|
|
2547
|
+
function ensureSpeechPhoneticChain(voiceoverAudioEntityId, audioScriptText, phonemeScript, entities, relations, addEntity, link, documentAudioScriptEntityId) {
|
|
2548
|
+
const renders = relations.filter((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voiceoverAudioEntityId);
|
|
2549
|
+
if (renders.length > 1) fail(`Voiceover ${voiceoverAudioEntityId} has ambiguous PhoneticScript provenance`);
|
|
2509
2550
|
if (renders.length === 1) {
|
|
2510
2551
|
const phonetic = entities.find((entity) => entity.entityId === renders[0].endpoint1EntityId && entity.entityKind === "phonetic-script");
|
|
2511
|
-
if (phonetic === void 0) fail(`
|
|
2552
|
+
if (phonetic === void 0) fail(`Voiceover ${voiceoverAudioEntityId} has a dangling PhoneticScript render`);
|
|
2512
2553
|
const script = composedAudioScript(phonetic, entities);
|
|
2513
|
-
if (script === void 0 || scriptText(script) !== audioScriptText) fail(`
|
|
2554
|
+
if (script === void 0 || scriptText(script) !== audioScriptText) fail(`Voiceover ${voiceoverAudioEntityId} conflicts with the legacy Speech script`);
|
|
2514
2555
|
return phonetic.entityId;
|
|
2515
2556
|
}
|
|
2516
2557
|
const script = entities.find((entity) => entity.entityId === documentAudioScriptEntityId);
|
|
2517
2558
|
if (script === void 0 || script.entityKind !== "audio-script") fail("Migration must carry the fixed document AudioScript");
|
|
2518
2559
|
const segments = script.payload.segments;
|
|
2519
2560
|
if (Array.isArray(segments) && segments.length > 0) {
|
|
2520
|
-
if (scriptText(script) !== audioScriptText) fail(`
|
|
2561
|
+
if (scriptText(script) !== audioScriptText) fail(`Voiceover ${voiceoverAudioEntityId} conflicts with the project AudioScript text`);
|
|
2521
2562
|
} else {
|
|
2522
2563
|
const scriptIndex = entities.findIndex((row) => row.entityId === script.entityId);
|
|
2523
2564
|
entities[scriptIndex] = {
|
|
@@ -2532,7 +2573,7 @@ function ensureSpeechPhoneticChain(voiceEntityId, audioScriptText, phonemeScript
|
|
|
2532
2573
|
baseEntityIds: [script.entityId],
|
|
2533
2574
|
...phonemeScript === void 0 ? {} : { phonemeScript }
|
|
2534
2575
|
});
|
|
2535
|
-
link("phonetic-script-render",
|
|
2576
|
+
link("phonetic-script-render", voiceoverAudioEntityId, phonetic);
|
|
2536
2577
|
return phonetic;
|
|
2537
2578
|
}
|
|
2538
2579
|
function composedAudioScript(phonetic, entities) {
|
package/dist/testing.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { St as bytesToBase64, o as createMirrorVideoDocumentAdapter, xt as base64ToBytes } from "./document-
|
|
1
|
+
import { St as bytesToBase64, o as createMirrorVideoDocumentAdapter, xt as base64ToBytes } from "./document-C30rnewk.js";
|
|
2
2
|
import { t as classifyUpdate } from "./loro-relay-doc-cJSY-uau.js";
|
|
3
3
|
import { LoroDoc, VersionVector, encodeFrontiers } from "loro-crdt";
|
|
4
4
|
//#region src/testing/in-memory-mengine-server.ts
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@mengine/medeo-client",
|
|
3
|
-
"version": "2.0.1-alpha.
|
|
3
|
+
"version": "2.0.1-alpha.9",
|
|
4
4
|
"license": "UNLICENSED",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -28,9 +28,9 @@
|
|
|
28
28
|
"immer": "^10.2.0",
|
|
29
29
|
"loro-mirror": "^2.3.3",
|
|
30
30
|
"zod": "^4.4.3",
|
|
31
|
-
"@mengine/storage": "2.0.1-alpha.
|
|
32
|
-
"@mengine/sync": "2.0.1-alpha.
|
|
33
|
-
"@mengine/utils": "2.0.1-alpha.
|
|
31
|
+
"@mengine/storage": "2.0.1-alpha.9",
|
|
32
|
+
"@mengine/sync": "2.0.1-alpha.9",
|
|
33
|
+
"@mengine/utils": "2.0.1-alpha.9"
|
|
34
34
|
},
|
|
35
35
|
"devDependencies": {
|
|
36
36
|
"@types/node": "^25.9.1",
|
|
@@ -40,8 +40,8 @@
|
|
|
40
40
|
"typescript": "^6.0.3",
|
|
41
41
|
"vite-plugin-wasm": "^3.6.0",
|
|
42
42
|
"vite-plus": "^0.1.23",
|
|
43
|
-
"@mengine/
|
|
44
|
-
"@mengine/
|
|
43
|
+
"@mengine/medeo-dsl": "0.0.0",
|
|
44
|
+
"@mengine/idl-codegen": "0.0.0"
|
|
45
45
|
},
|
|
46
46
|
"peerDependencies": {
|
|
47
47
|
"loro-crdt": "^1.16.1"
|