@mengine/medeo-client 2.0.1-alpha.7 → 2.0.1-alpha.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{document-Bnxm1p5E.js → document-DdwzfZ1g.js} +109 -30
- package/dist/index.d.ts +20 -10
- package/dist/index.js +102 -61
- package/dist/testing.js +1 -1
- package/package.json +4 -4
|
@@ -1447,8 +1447,63 @@ function emptyLaneTracks() {
|
|
|
1447
1447
|
}
|
|
1448
1448
|
//#endregion
|
|
1449
1449
|
//#region ../medeo-dsl/src/entities.ts
|
|
1450
|
+
/**
|
|
1451
|
+
* The declared trait set of every known kind.
|
|
1452
|
+
*
|
|
1453
|
+
* `caption` carries `sequence` on purpose: a Caption owns the display timing of
|
|
1454
|
+
* the one AudioScript segment it shows, which is what admits it into a Clip.
|
|
1455
|
+
* `voice` carries no media or sequence trait — it is a timbre identity, not
|
|
1456
|
+
* playable content; a rendered voiceover is an `audio` entity.
|
|
1457
|
+
*/
|
|
1458
|
+
const ENTITY_TRAIT_KINDS = Object.freeze({
|
|
1459
|
+
axvideo: Object.freeze([
|
|
1460
|
+
"container",
|
|
1461
|
+
"sequence",
|
|
1462
|
+
"visual",
|
|
1463
|
+
"audible"
|
|
1464
|
+
]),
|
|
1465
|
+
timeline: Object.freeze(["container"]),
|
|
1466
|
+
track: Object.freeze(["container", "sequence"]),
|
|
1467
|
+
clip: Object.freeze(["container"]),
|
|
1468
|
+
asset: Object.freeze([]),
|
|
1469
|
+
video: Object.freeze([
|
|
1470
|
+
"media-file",
|
|
1471
|
+
"sequence",
|
|
1472
|
+
"visual",
|
|
1473
|
+
"audible"
|
|
1474
|
+
]),
|
|
1475
|
+
audio: Object.freeze([
|
|
1476
|
+
"media-file",
|
|
1477
|
+
"sequence",
|
|
1478
|
+
"audible"
|
|
1479
|
+
]),
|
|
1480
|
+
image: Object.freeze([
|
|
1481
|
+
"media-file",
|
|
1482
|
+
"sequence",
|
|
1483
|
+
"visual"
|
|
1484
|
+
]),
|
|
1485
|
+
voice: Object.freeze(["container", "ip-bound"]),
|
|
1486
|
+
"sequence-marker": Object.freeze([]),
|
|
1487
|
+
viewport: Object.freeze([]),
|
|
1488
|
+
"audio-script": Object.freeze(["segment-text"]),
|
|
1489
|
+
"phonetic-script": Object.freeze(["segment-text"]),
|
|
1490
|
+
caption: Object.freeze(["segment-text", "sequence"])
|
|
1491
|
+
});
|
|
1492
|
+
/** Declared traits of a known kind; extension kinds declare none through this table. */
|
|
1493
|
+
function traitKindsOf(entityKind) {
|
|
1494
|
+
return isKnownEntityKind(entityKind) ? ENTITY_TRAIT_KINDS[entityKind] : [];
|
|
1495
|
+
}
|
|
1496
|
+
function hasTraitKind(entityKind, trait) {
|
|
1497
|
+
return traitKindsOf(entityKind).includes(trait);
|
|
1498
|
+
}
|
|
1499
|
+
/**
|
|
1500
|
+
* Sequence admission asks the declared traits first, so a kind that does not
|
|
1501
|
+
* declare `sequence` can never buy its way in by carrying the fields. Extension
|
|
1502
|
+
* kinds declare no traits and stay structurally detected.
|
|
1503
|
+
*/
|
|
1450
1504
|
function hasSequence(entity) {
|
|
1451
|
-
if (
|
|
1505
|
+
if (isReservedEntityKind(entity.entityKind)) return false;
|
|
1506
|
+
if (isKnownEntityKind(entity.entityKind) && !hasTraitKind(entity.entityKind, "sequence")) return false;
|
|
1452
1507
|
return isSequenceFields(entity);
|
|
1453
1508
|
}
|
|
1454
1509
|
function isSequenceFields(value) {
|
|
@@ -1461,22 +1516,16 @@ function isSequenceFields(value) {
|
|
|
1461
1516
|
if (value.sampling !== "native" && value.sampling !== "constant" && value.sampling !== "derived") return false;
|
|
1462
1517
|
return "coordinateSpace" in value && value.coordinateSpace !== void 0;
|
|
1463
1518
|
}
|
|
1464
|
-
function isKnownSequenceKind(kind) {
|
|
1465
|
-
return isMediaAssetVariantKind(kind) || kind === "caption" || kind === "axvideo";
|
|
1466
|
-
}
|
|
1467
1519
|
function isMediaAssetVariantKind(kind) {
|
|
1468
|
-
return kind === "video" || kind === "image" || kind === "audio"
|
|
1520
|
+
return kind === "video" || kind === "image" || kind === "audio";
|
|
1469
1521
|
}
|
|
1470
1522
|
function isKnownEntityKind(kind) {
|
|
1471
|
-
return
|
|
1523
|
+
return Object.hasOwn(ENTITY_TRAIT_KINDS, kind);
|
|
1472
1524
|
}
|
|
1473
1525
|
function isReservedEntityKind(kind) {
|
|
1474
1526
|
const normalized = kind.toLowerCase().replaceAll("-", "").replaceAll("_", "");
|
|
1475
1527
|
return normalized === "speech" || normalized === "videodocument";
|
|
1476
1528
|
}
|
|
1477
|
-
function isKnownNonSequenceKind(kind) {
|
|
1478
|
-
return kind === "timeline" || kind === "track" || kind === "clip" || kind === "asset" || kind === "sequence-marker" || kind === "viewport" || kind === "audio-script" || kind === "phonetic-script";
|
|
1479
|
-
}
|
|
1480
1529
|
function isRecord$1(value) {
|
|
1481
1530
|
return typeof value === "object" && value != null && !Array.isArray(value);
|
|
1482
1531
|
}
|
|
@@ -2059,12 +2108,15 @@ function validateKnownEntityPayload(entity) {
|
|
|
2059
2108
|
break;
|
|
2060
2109
|
case "video":
|
|
2061
2110
|
case "audio":
|
|
2062
|
-
|
|
2111
|
+
validateSequencePayload(value, "bounded", "native", problems);
|
|
2112
|
+
validateMediaAssetFields(value, problems);
|
|
2113
|
+
break;
|
|
2063
2114
|
case "caption":
|
|
2064
2115
|
validateSequencePayload(value, "bounded", "native", problems);
|
|
2065
|
-
|
|
2066
|
-
|
|
2067
|
-
|
|
2116
|
+
validateCaptionPayload(value, problems);
|
|
2117
|
+
break;
|
|
2118
|
+
case "voice":
|
|
2119
|
+
validateVoicePayload(value, problems);
|
|
2068
2120
|
break;
|
|
2069
2121
|
case "image":
|
|
2070
2122
|
validateSequencePayload(value, "unbounded", "constant", problems);
|
|
@@ -2171,11 +2223,26 @@ function validateExternalLocator(value, problems) {
|
|
|
2171
2223
|
function validateStorageKey(value, problems) {
|
|
2172
2224
|
if (value.storageKey !== void 0 && (typeof value.storageKey !== "string" || value.storageKey.trim() === "")) problems.push("storageKey must be a non-empty string when present");
|
|
2173
2225
|
}
|
|
2226
|
+
/**
|
|
2227
|
+
* Voice is a timbre identity, not playable content: it owns the voice-library
|
|
2228
|
+
* descriptor and nothing that would make it look like media or a Sequence. The
|
|
2229
|
+
* rendered voiceover is an `audio` entity bound by a `voice-timbre` Relation.
|
|
2230
|
+
*/
|
|
2174
2231
|
function validateVoicePayload(value, problems) {
|
|
2232
|
+
for (const key of [
|
|
2233
|
+
"extent",
|
|
2234
|
+
"sampling",
|
|
2235
|
+
"coordinateSpace",
|
|
2236
|
+
"external",
|
|
2237
|
+
"storageKey"
|
|
2238
|
+
]) if (Object.hasOwn(value, key)) problems.push(`${key} belongs to the rendered Audio, not to the Voice identity`);
|
|
2175
2239
|
const voice = value.voice;
|
|
2176
|
-
if (voice === void 0)
|
|
2240
|
+
if (voice === void 0) {
|
|
2241
|
+
problems.push("voice is required; a Voice entity is its voice-library identity");
|
|
2242
|
+
return;
|
|
2243
|
+
}
|
|
2177
2244
|
if (!isRecord(voice)) {
|
|
2178
|
-
problems.push("voice must be an object
|
|
2245
|
+
problems.push("voice must be an object");
|
|
2179
2246
|
return;
|
|
2180
2247
|
}
|
|
2181
2248
|
if (voice.system !== "voice-library") problems.push("voice.system must be \"voice-library\"");
|
|
@@ -2307,7 +2374,6 @@ function isOwnedLocalEntityValuePath(entityKind, path) {
|
|
|
2307
2374
|
if ([
|
|
2308
2375
|
"video",
|
|
2309
2376
|
"audio",
|
|
2310
|
-
"voice",
|
|
2311
2377
|
"image",
|
|
2312
2378
|
"caption",
|
|
2313
2379
|
"axvideo"
|
|
@@ -2371,7 +2437,7 @@ function collectRelatedEntities(entityRefs, index) {
|
|
|
2371
2437
|
};
|
|
2372
2438
|
}
|
|
2373
2439
|
function expectedSequenceShape(kind) {
|
|
2374
|
-
if (kind === "video" || kind === "audio" || kind === "
|
|
2440
|
+
if (kind === "video" || kind === "audio" || kind === "caption") return {
|
|
2375
2441
|
extent: "bounded",
|
|
2376
2442
|
sampling: "native"
|
|
2377
2443
|
};
|
|
@@ -2412,7 +2478,7 @@ const generatedRelationSpec = Object.freeze({
|
|
|
2412
2478
|
});
|
|
2413
2479
|
const captionAlignmentRelationSpec = Object.freeze({
|
|
2414
2480
|
kind: "caption-alignment",
|
|
2415
|
-
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["caption"]), new Set(["audio"
|
|
2481
|
+
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["caption"]), new Set(["audio"])),
|
|
2416
2482
|
validateMetadata: isCaptionAlignmentMetadata
|
|
2417
2483
|
});
|
|
2418
2484
|
/** `clip-anchor(child, host)` means endpoint 0 follows endpoint 1. */
|
|
@@ -2421,20 +2487,29 @@ const clipAnchorRelationSpec = Object.freeze({
|
|
|
2421
2487
|
validateEndpoints: (endpoints) => endpoints[0].current().entityKind === "clip" && endpoints[1].current().entityKind === "clip",
|
|
2422
2488
|
validateMetadata: isEmptyMetadata
|
|
2423
2489
|
});
|
|
2424
|
-
/**
|
|
2490
|
+
/**
|
|
2491
|
+
* The rendered voiceover Audio and the PhoneticScript it was synthesized from;
|
|
2492
|
+
* kinds determine roles regardless of endpoint positions.
|
|
2493
|
+
*/
|
|
2425
2494
|
const phoneticScriptRenderRelationSpec = Object.freeze({
|
|
2426
2495
|
kind: "phonetic-script-render",
|
|
2427
|
-
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["
|
|
2496
|
+
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio"]), new Set(["phonetic-script"])),
|
|
2497
|
+
validateMetadata: isEmptyMetadata
|
|
2498
|
+
});
|
|
2499
|
+
/**
|
|
2500
|
+
* The timbre identity a rendered voiceover Audio was synthesized with. Voice
|
|
2501
|
+
* stays an identity: it never carries the audio itself, so the link between the
|
|
2502
|
+
* two is a Relation rather than one shared entity.
|
|
2503
|
+
*/
|
|
2504
|
+
const voiceTimbreRelationSpec = Object.freeze({
|
|
2505
|
+
kind: "voice-timbre",
|
|
2506
|
+
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio"]), new Set(["voice"])),
|
|
2428
2507
|
validateMetadata: isEmptyMetadata
|
|
2429
2508
|
});
|
|
2430
2509
|
/** AudioScript was transcribed from Audio, Video, or recorded Voice; kinds determine roles. */
|
|
2431
2510
|
const audioScriptSourceRelationSpec = Object.freeze({
|
|
2432
2511
|
kind: "audio-script-source",
|
|
2433
|
-
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio-script"]), new Set([
|
|
2434
|
-
"audio",
|
|
2435
|
-
"video",
|
|
2436
|
-
"voice"
|
|
2437
|
-
])),
|
|
2512
|
+
validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio-script"]), new Set(["audio", "video"])),
|
|
2438
2513
|
validateMetadata: isEmptyMetadata
|
|
2439
2514
|
});
|
|
2440
2515
|
/**
|
|
@@ -2461,6 +2536,7 @@ const builtInRelationSpecs = Object.freeze([
|
|
|
2461
2536
|
captionAlignmentRelationSpec,
|
|
2462
2537
|
clipAnchorRelationSpec,
|
|
2463
2538
|
phoneticScriptRenderRelationSpec,
|
|
2539
|
+
voiceTimbreRelationSpec,
|
|
2464
2540
|
audioScriptSourceRelationSpec,
|
|
2465
2541
|
audioScriptMarkerRelationSpec
|
|
2466
2542
|
]);
|
|
@@ -2490,7 +2566,7 @@ function hasAssetAndSequence(endpoints) {
|
|
|
2490
2566
|
return first.entityKind === "asset" && hasSequence(second) || second.entityKind === "asset" && hasSequence(first);
|
|
2491
2567
|
}
|
|
2492
2568
|
function isGeneratedMedia(entity) {
|
|
2493
|
-
return entity.entityKind === "video" || entity.entityKind === "image" || entity.entityKind === "audio"
|
|
2569
|
+
return entity.entityKind === "video" || entity.entityKind === "image" || entity.entityKind === "audio";
|
|
2494
2570
|
}
|
|
2495
2571
|
function isEmptyMetadata(value) {
|
|
2496
2572
|
return isJsonObject(value) && Object.keys(value).length === 0;
|
|
@@ -2920,7 +2996,7 @@ function resolveGraphLayout(graph) {
|
|
|
2920
2996
|
if (clip.entityKind !== "clip" || graph.adjacent(clip.entityId, "track-clip").length !== 1) fail(`Clip ${clip.entityId} must belong to exactly one Track`);
|
|
2921
2997
|
const marker = graph.one(clip.entityId, "clip-marker", "sequence-marker");
|
|
2922
2998
|
const content = graph.one(marker.entityId, "marker-content");
|
|
2923
|
-
if (!(typedRole === "video_clip" ? ["image", "video"] : typedRole === "speech" ? ["
|
|
2999
|
+
if (!(typedRole === "video_clip" ? ["image", "video"] : typedRole === "speech" ? ["audio"] : typedRole === "caption" ? ["caption"] : ["audio"]).includes(content.entityKind)) fail(`unsupported compatibility content kind ${content.entityKind} on ${typedRole} Track`);
|
|
2924
3000
|
const space = content.payload.coordinateSpace;
|
|
2925
3001
|
if (space !== "ms" && space !== "milliseconds" && (!object(space) || space.unit !== "ms" && space.unit !== "milliseconds")) fail(`media ${content.entityId} must state millisecond coordinates for this reader`);
|
|
2926
3002
|
const source = range(marker.payload.sourceRange, "sourceRange");
|
|
@@ -2999,7 +3075,7 @@ function resolveGraphLayout(graph) {
|
|
|
2999
3075
|
let start = fact.absoluteStart ?? 0;
|
|
3000
3076
|
if (fact.mode === "anchored") {
|
|
3001
3077
|
const host = resolve(fact.anchorId);
|
|
3002
|
-
if (fact.role === "speech" && !["video_clip", "bgm"].includes(facts.get(host.clipEntityId).role)) fail("An anchored
|
|
3078
|
+
if (fact.role === "speech" && !["video_clip", "bgm"].includes(facts.get(host.clipEntityId).role)) fail("An anchored voiceover must use a visual or BGM Clip as its host");
|
|
3003
3079
|
start = host.start + fact.anchorOffset;
|
|
3004
3080
|
} else if (fact.predecessor !== void 0) {
|
|
3005
3081
|
const prior = resolve(fact.predecessor);
|
|
@@ -3083,8 +3159,11 @@ function projectEntityTimeline(rows) {
|
|
|
3083
3159
|
const asset = mediaAssetFacts(graph, content);
|
|
3084
3160
|
requireUntrimmedAudio(content, marker, placement);
|
|
3085
3161
|
const external = externalLocator(asset, "memota-speech");
|
|
3086
|
-
const
|
|
3087
|
-
const
|
|
3162
|
+
const renderEdges = graph.adjacent(content.entityId, "phonetic-script-render");
|
|
3163
|
+
const timbreEdges = graph.adjacent(content.entityId, "voice-timbre");
|
|
3164
|
+
const assembled = renderEdges.length > 0 || timbreEdges.length > 0 ? assemblePhoneticScriptContent(rows, graph.one(content.entityId, "phonetic-script-render", "phonetic-script").entityId) : void 0;
|
|
3165
|
+
if (timbreEdges.length > 1) fail("A voiceover resolves at most one Voice identity");
|
|
3166
|
+
const voice = timbreEdges.length === 0 ? void 0 : graph.one(content.entityId, "voice-timbre", "voice").payload.voice;
|
|
3088
3167
|
if (voice !== void 0 && (!object(voice) || voice.system !== "voice-library" || typeof voice.key !== "string")) fail("Voice must use an explicit voice-library locator");
|
|
3089
3168
|
const captionIds = layout.tracks.flatMap((candidate) => candidate.role === "caption" ? candidate.clips.filter((caption) => caption.anchorClipEntityId === clip.entityId).map((caption) => caption.clipEntityId) : []);
|
|
3090
3169
|
parts[clip.entityId] = { speech: {
|
package/dist/index.d.ts
CHANGED
|
@@ -272,17 +272,22 @@ interface AudioMediaAssetFact {
|
|
|
272
272
|
readonly durationMs: number;
|
|
273
273
|
readonly storageKey: string;
|
|
274
274
|
}
|
|
275
|
-
|
|
275
|
+
/**
|
|
276
|
+
* A speech recording. It materializes as an `audio` entity like every other
|
|
277
|
+
* playable sound; `voice` names the timbre identity it was synthesized with,
|
|
278
|
+
* which becomes its own Voice entity linked by a `voice-timbre` Relation.
|
|
279
|
+
*/
|
|
280
|
+
interface VoiceoverMediaAssetFact {
|
|
276
281
|
/** Stable external speech result id, independent of the placed Clip id. */
|
|
277
282
|
readonly assetId: string;
|
|
278
|
-
readonly kind: '
|
|
283
|
+
readonly kind: 'voiceover';
|
|
279
284
|
readonly durationMs: number;
|
|
280
285
|
readonly storageKey: string;
|
|
281
|
-
/** Present for synthesized
|
|
286
|
+
/** Present for synthesized speech, absent for original recorded audio. */
|
|
282
287
|
readonly voice?: VoiceDescriptor;
|
|
283
288
|
}
|
|
284
289
|
/** Facts resolved from media storage. A trim window never substitutes for intrinsic duration. */
|
|
285
|
-
type MediaAssetFact = ImageMediaAssetFact | VideoMediaAssetFact | AudioMediaAssetFact |
|
|
290
|
+
type MediaAssetFact = ImageMediaAssetFact | VideoMediaAssetFact | AudioMediaAssetFact | VoiceoverMediaAssetFact;
|
|
286
291
|
type ClipPlacement = {
|
|
287
292
|
readonly kind: 'sequential';
|
|
288
293
|
readonly order: number;
|
|
@@ -334,7 +339,7 @@ interface DeleteClipTreeInput {
|
|
|
334
339
|
interface VoiceoverCaptionFact {
|
|
335
340
|
/** Stable placed caption identity supplied by the materialized side effect. */
|
|
336
341
|
readonly captionClipEntityId: string;
|
|
337
|
-
/** Directly held bases; includes the AudioScript
|
|
342
|
+
/** Directly held bases; includes the AudioScript the voiceover was rendered from. */
|
|
338
343
|
readonly baseEntityIds: readonly string[];
|
|
339
344
|
/** Ordered selection of AudioScript segments; caption text is never passed inline. */
|
|
340
345
|
readonly selection: CaptionSegmentSelection;
|
|
@@ -345,7 +350,7 @@ interface VoiceoverCaptionFact {
|
|
|
345
350
|
type VoiceoverTakeInput = {
|
|
346
351
|
readonly timelineEntityId: string; /** Stable placed speech identity, distinct from media.assetId. */
|
|
347
352
|
readonly voiceoverClipEntityId: string;
|
|
348
|
-
readonly media:
|
|
353
|
+
readonly media: VoiceoverMediaAssetFact; /** Existing pronunciation variant; its composed AudioScript stays the text owner. */
|
|
349
354
|
readonly phoneticScriptEntityId: string;
|
|
350
355
|
readonly volume: number;
|
|
351
356
|
readonly captions: readonly VoiceoverCaptionFact[];
|
|
@@ -421,8 +426,11 @@ interface TrimClipInput {
|
|
|
421
426
|
}
|
|
422
427
|
interface VoiceoverTakeResult {
|
|
423
428
|
readonly voiceoverClipEntityId: string;
|
|
424
|
-
|
|
425
|
-
|
|
429
|
+
/** The rendered voiceover Audio entity. */
|
|
430
|
+
readonly voiceoverAudioEntityId: string;
|
|
431
|
+
/** The timbre identity it was synthesized with, when the take declared one. */
|
|
432
|
+
readonly voiceEntityId?: string;
|
|
433
|
+
/** The pronunciation variant the voiceover was rendered from. */
|
|
426
434
|
readonly phoneticScriptEntityId: string;
|
|
427
435
|
/** The base-text owner resolved from the PhoneticScript baseEntityIds. */
|
|
428
436
|
readonly audioScriptEntityId: string;
|
|
@@ -556,7 +564,7 @@ declare class EntityTimelineEditor {
|
|
|
556
564
|
insertCaptionClip(input: InsertCaptionClipInput): ClipEntityId;
|
|
557
565
|
private insertPlacedClipIntoDraft;
|
|
558
566
|
private ensureRoleTrack;
|
|
559
|
-
/** The existing pronunciation variant a
|
|
567
|
+
/** The existing pronunciation variant a voiceover consumes; composition is required. */
|
|
560
568
|
private requireComposedPhoneticScript;
|
|
561
569
|
/** Preserve the factual synthesis input, regardless of persisted endpoint positions. */
|
|
562
570
|
private ensurePhoneticRender;
|
|
@@ -632,6 +640,8 @@ interface ImportedMediaAsset {
|
|
|
632
640
|
readonly rows: EntityRelationRows;
|
|
633
641
|
/** The single media variant identity for this asset id; it is its own Asset. */
|
|
634
642
|
readonly contentEntityId: string;
|
|
643
|
+
/** The timbre identity of a synthesized voiceover; absent for recorded media. */
|
|
644
|
+
readonly voiceEntityId?: string;
|
|
635
645
|
}
|
|
636
646
|
/**
|
|
637
647
|
* Import at the editor boundary, not at asset generation time. Existing
|
|
@@ -1716,4 +1726,4 @@ declare function hasAudioScriptAsset(payload: JsonObject): boolean;
|
|
|
1716
1726
|
/** Internal fingerprint prevents a resource from silently describing an older text revision. */
|
|
1717
1727
|
declare function audioScriptAssetFields(payload: JsonObject, assetId: string): JsonObject;
|
|
1718
1728
|
//#endregion
|
|
1719
|
-
export { type Aggregation, type AssembledCaptionContent, type AssembledPhoneticScriptContent, type Attachment, type AudioMediaAssetFact, type BgmPart, type CaptionContentInput, type CaptionDisplayCue, type CaptionPart, type CaptionSegmentSelection, type CaptionStyle, type ClipEntityId, type ClipPlacement, type CommitOptions, DEFAULT_UNIT_TIME_MS, type DeleteBgmInput, type DeleteClipInput, type DeleteClipTreeInput, type DeleteVoiceoverInput, type DerivedItemPosition, type DocStorageLike, type DocVersionMark, type DocumentAudioScript, type DocumentAudioScriptState, type EditorFoundation, EntityDocumentState, type EntityGraphCommitOptions, EntityGraphHttpClient, type EntityGraphState, EntityTimelineEditor, type EntityTimelineIdFactory, EntityTimelineProjectionError, type EntityTimelineTrackRole, type FieldChange, type FieldPath, IMPLEMENTED_SEMANTIC_OP_KINDS, type ImageMediaAssetFact, type ImplementedSemanticOpKind, type ImportedMediaAsset, type InsertCaptionClipInput, type InsertClipInput, type InsertMediaClipInput, type InsertMediaClipsInput, type InsertPlacedClipInput, type JournalEntry, LANE_KINDS_IN_STACK_ORDER, LORO_ENTITY_SCHEMA, type LaneKind, type LinearClipSpeed, LoroEntityDocument, LoroEntityDraft, type MainClipRange, type MakeEmptyPart, ManualSyncDoc, type ManualSyncDocOptions, MedeoHttpDocStorage, type MedeoHttpDocStorageOptions, type MediaAssetFact, type MediaClipInsertion, MemoryDocStorage, MengineAckFailedError, type MengineAuditEntry, type MengineAuditResponse, MengineDocSession, type MengineDocSessionOptions, type MengineDocSessionUpdateEvent, type MengineDocSyncState, type MengineDocumentVersion, type MengineEventStreamOptions, MengineHttpClient, type MengineHttpClientOptions, MengineHttpRequestError, MenginePushRejectedError, type MenginePushResponse, type MenginePushUpdateResponse, type MengineRejectedResponse, type MengineSnapshotResponse, type MengineSseUpdateEvent, type MengineSyncResponse, type MengineUndoState, type MengineUpdateMeta, MirrorVideoDocumentAdapter, type MirrorVideoDocumentOptions, type MoveClipInput, type MoveClipsToStartsInput, type MoveSequentialClipsInput, type MoveVoiceoverInput, type OpActor, type PartAggregation, type PartIdFactory, type PartKind, type PartUnion, type PatchCaptionStyleInput, PlainMemoryAdapter, type PlainMemoryAdapterOptions, type PlannedSemanticOpKind, type PullFailureReason, type PullResult, type PushOutcome, type PushOutcomeKind, type PushResult, type PushResultKind, type ReplaceClipContentInput, type ReplaceMediaClipInput, type ReplaceSequentialClipsInput, type ReplacementMediaClipInput, type ResolvedEntityClipPlacement, type ResolvedEntityTimelineLayout, type ResolvedEntityTrack, SchemaValidator, ScriptCompositionError, type SemanticDocumentAdapter, SemanticEditor, type SemanticOpInput, type SemanticOpKind, type SemanticOpName, type SequentialClipAnchor, type SetBgmInput, type SetCaptionVisibilityInput, type SetClipPlacementInput, type SetClipSpeedInput, type SetClipVolumeInput, type SnapshotReadable, type SolvedVideoDocument, type SpeechHostMap, type SpeechPart, type SpeedShift, TIMELINE_SKELETON_DURATION_MS, type Timeline, type TimelineDoc, type TimelineItem, type Track, type TrackDraft, type TrackItem, type TrackItemDraft, type TrackItemTimePosition, type TransactAudit, type TrimClipInput, type UpdateClipInput, type UpdateClipMarkerInput, ValidationError, type VideoClipPart, type VideoDocument, type VideoDocumentDraft, type VideoDocumentMirrorSchema, VideoDocumentValidationError, type VideoDocumentValidationIssue, type VideoDocumentValidationIssueCode, type VideoDraft, type CaptionPart$1 as VideoDraftCaptionPart, type VideoDraftContent, type VideoDraftPartUnion, type Timeline$1 as VideoDraftTimeline, type Track$1 as VideoDraftTrack, type TrackItem$1 as VideoDraftTrackItem, type VideoMediaAssetFact, type VisualMediaAssetFact, type
|
|
1729
|
+
export { type Aggregation, type AssembledCaptionContent, type AssembledPhoneticScriptContent, type Attachment, type AudioMediaAssetFact, type BgmPart, type CaptionContentInput, type CaptionDisplayCue, type CaptionPart, type CaptionSegmentSelection, type CaptionStyle, type ClipEntityId, type ClipPlacement, type CommitOptions, DEFAULT_UNIT_TIME_MS, type DeleteBgmInput, type DeleteClipInput, type DeleteClipTreeInput, type DeleteVoiceoverInput, type DerivedItemPosition, type DocStorageLike, type DocVersionMark, type DocumentAudioScript, type DocumentAudioScriptState, type EditorFoundation, EntityDocumentState, type EntityGraphCommitOptions, EntityGraphHttpClient, type EntityGraphState, EntityTimelineEditor, type EntityTimelineIdFactory, EntityTimelineProjectionError, type EntityTimelineTrackRole, type FieldChange, type FieldPath, IMPLEMENTED_SEMANTIC_OP_KINDS, type ImageMediaAssetFact, type ImplementedSemanticOpKind, type ImportedMediaAsset, type InsertCaptionClipInput, type InsertClipInput, type InsertMediaClipInput, type InsertMediaClipsInput, type InsertPlacedClipInput, type JournalEntry, LANE_KINDS_IN_STACK_ORDER, LORO_ENTITY_SCHEMA, type LaneKind, type LinearClipSpeed, LoroEntityDocument, LoroEntityDraft, type MainClipRange, type MakeEmptyPart, ManualSyncDoc, type ManualSyncDocOptions, MedeoHttpDocStorage, type MedeoHttpDocStorageOptions, type MediaAssetFact, type MediaClipInsertion, MemoryDocStorage, MengineAckFailedError, type MengineAuditEntry, type MengineAuditResponse, MengineDocSession, type MengineDocSessionOptions, type MengineDocSessionUpdateEvent, type MengineDocSyncState, type MengineDocumentVersion, type MengineEventStreamOptions, MengineHttpClient, type MengineHttpClientOptions, MengineHttpRequestError, MenginePushRejectedError, type MenginePushResponse, type MenginePushUpdateResponse, type MengineRejectedResponse, type MengineSnapshotResponse, type MengineSseUpdateEvent, type MengineSyncResponse, type MengineUndoState, type MengineUpdateMeta, MirrorVideoDocumentAdapter, type MirrorVideoDocumentOptions, type MoveClipInput, type MoveClipsToStartsInput, type MoveSequentialClipsInput, type MoveVoiceoverInput, type OpActor, type PartAggregation, type PartIdFactory, type PartKind, type PartUnion, type PatchCaptionStyleInput, PlainMemoryAdapter, type PlainMemoryAdapterOptions, type PlannedSemanticOpKind, type PullFailureReason, type PullResult, type PushOutcome, type PushOutcomeKind, type PushResult, type PushResultKind, type ReplaceClipContentInput, type ReplaceMediaClipInput, type ReplaceSequentialClipsInput, type ReplacementMediaClipInput, type ResolvedEntityClipPlacement, type ResolvedEntityTimelineLayout, type ResolvedEntityTrack, SchemaValidator, ScriptCompositionError, type SemanticDocumentAdapter, SemanticEditor, type SemanticOpInput, type SemanticOpKind, type SemanticOpName, type SequentialClipAnchor, type SetBgmInput, type SetCaptionVisibilityInput, type SetClipPlacementInput, type SetClipSpeedInput, type SetClipVolumeInput, type SnapshotReadable, type SolvedVideoDocument, type SpeechHostMap, type SpeechPart, type SpeedShift, TIMELINE_SKELETON_DURATION_MS, type Timeline, type TimelineDoc, type TimelineItem, type Track, type TrackDraft, type TrackItem, type TrackItemDraft, type TrackItemTimePosition, type TransactAudit, type TrimClipInput, type UpdateClipInput, type UpdateClipMarkerInput, ValidationError, type VideoClipPart, type VideoDocument, type VideoDocumentDraft, type VideoDocumentMirrorSchema, VideoDocumentValidationError, type VideoDocumentValidationIssue, type VideoDocumentValidationIssueCode, type VideoDraft, type CaptionPart$1 as VideoDraftCaptionPart, type VideoDraftContent, type VideoDraftPartUnion, type Timeline$1 as VideoDraftTimeline, type Track$1 as VideoDraftTrack, type TrackItem$1 as VideoDraftTrackItem, type VideoMediaAssetFact, type VisualMediaAssetFact, type VoiceoverCaptionFact, type VoiceoverMediaAssetFact, type VoiceoverTakeInput, type VoiceoverTakeResult, type WaitForServerAckOptions, applyEntityRows, applyFieldChanges, arrangeMainTrackSeamlessly, assembleCaptionContent, assembleEntityContent, assemblePhoneticScriptContent, assembleScriptText, assertCanonicalEditorResources, assertEntityProjectionPreservesLegacy, assertFieldChanges, assertMediaAssetWritePolicy, assertValidVideoDocument, audioScriptAssetContent, audioScriptAssetFields, audioScriptAssetHash, base64ToBytes, buildInitialVideoDocument, buildSpeechHostMap, bytesToBase64, cascadeAfterVideoClipChanges, compileEntityRows, createEditSandbox, createMirrorVideoDocument, createMirrorVideoDocumentAdapter, createPlainMemoryAdapter, decodeDocVersionMark, derivePositionFromAbs, effectiveVideoClipDurationMs, encodeDocVersionMark, ensureEditorFoundation, ensureLaneTrack, fillMainTrackTimeGaps, findComposedAudioScript, findLaneTrack, fromVideoDocument, generatePartId, getAt, hasAudioScriptAsset, hostForAbsMs, importMediaAsset, isEmptyVideoClip, isImplementedSemanticOpKind, isMap, laneTrackId, mainTrackRanges, migrateLegacyTimelineToEntities, partDurationMs, partUnionSchema, prepareCaptionContent, projectEntityTimeline, readDocumentAudioScript, readEntityDocument, readMainTrackItems, readMengineEventStream, readPart, readPartDurationMs, readVideoDocumentFromDraft, reassignSpeechesToVideoClipsByTime, recalculateTimelineDuration, relativePositionForAbs, replayJournal, resolveAllSpeechOverlaps, resolveEntityTimelineLayout, resolveFieldPath, resolveSpeechOverlapByShiftingVideos, safeDurationMs, index_d_exports as schemas, selectAudioScriptSegment, snapshotToPlain, solveVideoDocument, speedOf, syncAggregatedClipsTimePosition, toVideoDocument, validateVideoDocument, videoDocumentMirrorSchema, videoDocumentSchema };
|
package/dist/index.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { $ as findLaneTrack, A as assemblePhoneticScriptContent, B as isMediaAssetVariantKind, C as EntityProjectionGraph, D as ScriptCompositionError, E as decodeEntityRelationRows, F as variantBaseEntityIds, G as toVideoDocument, H as buildSpeechHostMap, I as isJsonObject, J as validateVideoDocument, K as VideoDocumentValidationError, L as createEntityId, M as findComposedAudioScript, N as selectAudioScriptSegment, O as assembleCaptionContent, P as updateEntityFields, Q as ensureLaneTrack, R as createRelationId, S as projectEntityTimeline, St as bytesToBase64, T as EntityTimelineProjectionError, U as derivePositionFromAbs, V as buildInitialVideoDocument, W as fromVideoDocument, X as videoDocumentSchema, Y as partUnionSchema, Z as LANE_KINDS_IN_STACK_ORDER, _ as hasAudioScriptAsset, _t as speedOf, a as createMirrorVideoDocument, at as fillMainTrackTimeGaps, b as assertFieldChanges, bt as videoDocumentMirrorSchema, c as DocumentMutationGuard, ct as resolveAllSpeechOverlaps, d as LoroEntityDraft, dt as TIMELINE_SKELETON_DURATION_MS, et as hasResolvedPlacement, f as readEntityDocument, ft as isEmptyVideoClip, g as audioScriptAssetHash, gt as effectiveVideoClipDurationMs, h as audioScriptAssetFields, ht as DEFAULT_UNIT_TIME_MS, i as MirrorVideoDocumentAdapter, it as arrangeMainTrackSeamlessly, j as assembleScriptText, k as assembleEntityContent, l as LORO_ENTITY_SCHEMA, lt as resolveSpeechOverlapByShiftingVideos, m as audioScriptAssetContent, mt as safeDurationMs, n as createPlainMemoryAdapter, nt as solveVideoDocument, o as createMirrorVideoDocumentAdapter, ot as reassignSpeechesToVideoClipsByTime, p as entityOrderGroups, pt as partDurationMs, q as assertValidVideoDocument, r as generatePartId, rt as cascadeAfterVideoClipChanges, s as readVideoDocumentFromDraft, st as recalculateTimelineDuration, t as PlainMemoryAdapter, tt as laneTrackId, u as LoroEntityDocument, ut as syncAggregatedClipsTimePosition, v as equalJson, vt as partUnionToDraft, w as resolveEntityTimelineLayout, x as resolveFieldPath, xt as base64ToBytes, y as applyFieldChanges, yt as recordEntries, z as hasSequence } from "./document-
|
|
1
|
+
import { $ as findLaneTrack, A as assemblePhoneticScriptContent, B as isMediaAssetVariantKind, C as EntityProjectionGraph, D as ScriptCompositionError, E as decodeEntityRelationRows, F as variantBaseEntityIds, G as toVideoDocument, H as buildSpeechHostMap, I as isJsonObject, J as validateVideoDocument, K as VideoDocumentValidationError, L as createEntityId, M as findComposedAudioScript, N as selectAudioScriptSegment, O as assembleCaptionContent, P as updateEntityFields, Q as ensureLaneTrack, R as createRelationId, S as projectEntityTimeline, St as bytesToBase64, T as EntityTimelineProjectionError, U as derivePositionFromAbs, V as buildInitialVideoDocument, W as fromVideoDocument, X as videoDocumentSchema, Y as partUnionSchema, Z as LANE_KINDS_IN_STACK_ORDER, _ as hasAudioScriptAsset, _t as speedOf, a as createMirrorVideoDocument, at as fillMainTrackTimeGaps, b as assertFieldChanges, bt as videoDocumentMirrorSchema, c as DocumentMutationGuard, ct as resolveAllSpeechOverlaps, d as LoroEntityDraft, dt as TIMELINE_SKELETON_DURATION_MS, et as hasResolvedPlacement, f as readEntityDocument, ft as isEmptyVideoClip, g as audioScriptAssetHash, gt as effectiveVideoClipDurationMs, h as audioScriptAssetFields, ht as DEFAULT_UNIT_TIME_MS, i as MirrorVideoDocumentAdapter, it as arrangeMainTrackSeamlessly, j as assembleScriptText, k as assembleEntityContent, l as LORO_ENTITY_SCHEMA, lt as resolveSpeechOverlapByShiftingVideos, m as audioScriptAssetContent, mt as safeDurationMs, n as createPlainMemoryAdapter, nt as solveVideoDocument, o as createMirrorVideoDocumentAdapter, ot as reassignSpeechesToVideoClipsByTime, p as entityOrderGroups, pt as partDurationMs, q as assertValidVideoDocument, r as generatePartId, rt as cascadeAfterVideoClipChanges, s as readVideoDocumentFromDraft, st as recalculateTimelineDuration, t as PlainMemoryAdapter, tt as laneTrackId, u as LoroEntityDocument, ut as syncAggregatedClipsTimePosition, v as equalJson, vt as partUnionToDraft, w as resolveEntityTimelineLayout, x as resolveFieldPath, xt as base64ToBytes, y as applyFieldChanges, yt as recordEntries, z as hasSequence } from "./document-DdwzfZ1g.js";
|
|
2
2
|
import { addSpeechesInputSchema, addVideoClipsInputSchema, adjustBgmVolumeInputSchema, adjustSpeechVolumeInputSchema, adjustVideoClipDurationInputSchema, adjustVideoClipVolumeInputSchema, changeSpeechScriptInputSchema, changeSpeechVoiceInputSchema, deleteBgmInputSchema, deleteSpeechesInputSchema, deleteVideoClipsInputSchema, moveSpeechesInputSchema, moveVideoClipsByAnchorInputSchema, moveVideoClipsInputSchema, replaceVideoClipContentInputSchema, replaceVideoClipSequenceInputSchema, setBgmInputSchema, setCaptionStyleInputSchema, setCaptionVisibilityInputSchema, setVideoClipSpeedShiftInputSchema, t as schemas_exports } from "./schemas.js";
|
|
3
3
|
import { LoroDoc, UndoManager, VersionVector } from "loro-crdt";
|
|
4
4
|
import { ClientServerSynchronizer, DocManager } from "@mengine/sync";
|
|
@@ -579,36 +579,55 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
|
|
|
579
579
|
"image",
|
|
580
580
|
"video",
|
|
581
581
|
"audio",
|
|
582
|
-
"
|
|
582
|
+
"voiceover"
|
|
583
583
|
].includes(fact.kind)) throw new Error("Unsupported media kind");
|
|
584
584
|
if (fact.kind !== "image" && (!Number.isFinite(fact.durationMs) || fact.durationMs <= 0)) throw new Error(`${fact.kind} ${fact.assetId} requires its factual positive duration`);
|
|
585
|
-
if ((fact.kind === "audio" || fact.kind === "
|
|
585
|
+
if ((fact.kind === "audio" || fact.kind === "voiceover") && fact.storageKey.trim() === "") throw new Error(`${fact.kind} ${fact.assetId} requires its physical storage key`);
|
|
586
586
|
decodeEntityRelationRows(initial, numericCoordinates);
|
|
587
587
|
assertCanonicalEditorResources(initial);
|
|
588
588
|
const entities = structuredClone([...initial.entities]);
|
|
589
589
|
const relations = structuredClone([...initial.relations]);
|
|
590
|
-
const system = fact.kind === "
|
|
590
|
+
const system = fact.kind === "voiceover" ? "memota-speech" : "memota";
|
|
591
|
+
const entityKind = fact.kind === "voiceover" ? "audio" : fact.kind;
|
|
592
|
+
const timbre = fact.kind === "voiceover" ? fact.voice : void 0;
|
|
591
593
|
const existing = entities.filter((row) => {
|
|
592
594
|
const external = row.payload.external;
|
|
593
595
|
return isMediaAssetVariantKind(row.entityKind) && isJsonObject(external) && external.system === system && external.key === fact.assetId;
|
|
594
596
|
});
|
|
597
|
+
const requireMatchingVariant = (row) => {
|
|
598
|
+
const extent = row.payload.extent;
|
|
599
|
+
if (row.entityKind !== entityKind || fact.kind !== "image" && (!isJsonObject(extent) || extent.start !== 0 || extent.end !== fact.durationMs)) throw new Error(`Media facts conflict with existing identity for ${fact.assetId}`);
|
|
600
|
+
if (fact.storageKey !== void 0 && row.payload.storageKey !== fact.storageKey) {
|
|
601
|
+
if (row.payload.storageKey !== void 0) throw new Error(`Media facts conflict with existing storage key for ${fact.assetId}`);
|
|
602
|
+
return {
|
|
603
|
+
...row,
|
|
604
|
+
payload: {
|
|
605
|
+
...row.payload,
|
|
606
|
+
storageKey: fact.storageKey
|
|
607
|
+
}
|
|
608
|
+
};
|
|
609
|
+
}
|
|
610
|
+
return row;
|
|
611
|
+
};
|
|
595
612
|
if (existing.length > 1) throw new Error(`Duplicate media identities for ${fact.assetId}; reuse ${existing[0].entityId}`);
|
|
596
613
|
if (existing[0] !== void 0) {
|
|
597
|
-
const variant = requireMatchingVariant(existing[0]
|
|
614
|
+
const variant = requireMatchingVariant(existing[0]);
|
|
598
615
|
replaceEntity$1(entities, variant.entityId, variant);
|
|
616
|
+
const voiceEntityId = bindVoiceTimbre(entities, relations, variant.entityId, timbre, idFactory);
|
|
599
617
|
return {
|
|
600
618
|
rows: {
|
|
601
619
|
entities,
|
|
602
620
|
relations
|
|
603
621
|
},
|
|
604
|
-
contentEntityId: variant.entityId
|
|
622
|
+
contentEntityId: variant.entityId,
|
|
623
|
+
...voiceEntityId
|
|
605
624
|
};
|
|
606
625
|
}
|
|
607
626
|
const entityId = createEntityId(idFactory("entity"));
|
|
608
627
|
if (entities.some((row) => row.entityId === entityId)) throw new Error(`Duplicate entity id ${entityId}`);
|
|
609
628
|
const variant = {
|
|
610
629
|
entityId,
|
|
611
|
-
entityKind
|
|
630
|
+
entityKind,
|
|
612
631
|
payload: {
|
|
613
632
|
extent: fact.kind === "image" ? {
|
|
614
633
|
kind: "unbounded",
|
|
@@ -624,15 +643,11 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
|
|
|
624
643
|
system,
|
|
625
644
|
key: fact.assetId
|
|
626
645
|
},
|
|
627
|
-
...fact.storageKey === void 0 ? {} : { storageKey: fact.storageKey }
|
|
628
|
-
...fact.kind === "voice" && fact.voice !== void 0 ? { voice: {
|
|
629
|
-
system: fact.voice.system,
|
|
630
|
-
key: fact.voice.key,
|
|
631
|
-
...fact.voice.name === void 0 ? {} : { name: fact.voice.name }
|
|
632
|
-
} } : {}
|
|
646
|
+
...fact.storageKey === void 0 ? {} : { storageKey: fact.storageKey }
|
|
633
647
|
}
|
|
634
648
|
};
|
|
635
649
|
entities.push(variant);
|
|
650
|
+
const voiceEntityId = bindVoiceTimbre(entities, relations, entityId, timbre, idFactory);
|
|
636
651
|
const rows = {
|
|
637
652
|
entities,
|
|
638
653
|
relations
|
|
@@ -641,33 +656,58 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
|
|
|
641
656
|
assertCanonicalEditorResources(rows);
|
|
642
657
|
return {
|
|
643
658
|
rows,
|
|
644
|
-
contentEntityId: entityId
|
|
659
|
+
contentEntityId: entityId,
|
|
660
|
+
...voiceEntityId
|
|
645
661
|
};
|
|
646
662
|
}
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
};
|
|
663
|
+
/**
|
|
664
|
+
* Attach the timbre a synthesized voiceover was rendered with. One Voice
|
|
665
|
+
* identity is shared by every take naming the same library entry, so an
|
|
666
|
+
* existing row is reused rather than duplicated. Recorded media declares no
|
|
667
|
+
* timbre and keeps no link.
|
|
668
|
+
*/
|
|
669
|
+
function bindVoiceTimbre(entities, relations, audioEntityId, descriptor, idFactory) {
|
|
670
|
+
const linked = relations.filter((relation) => relation.relationKind === "voice-timbre" && (relation.endpoint0EntityId === audioEntityId || relation.endpoint1EntityId === audioEntityId));
|
|
671
|
+
if (linked.length > 1) throw new Error(`Voiceover ${audioEntityId} has multiple Voice timbre relations`);
|
|
672
|
+
if (descriptor === void 0) {
|
|
673
|
+
if (linked.length > 0) throw new Error(`Recorded media ${audioEntityId} cannot keep a synthesized Voice timbre`);
|
|
674
|
+
return {};
|
|
659
675
|
}
|
|
660
|
-
|
|
676
|
+
const voice = entities.find((row) => row.entityKind === "voice" && sameVoiceDescriptor(row.payload.voice, descriptor)) ?? appendVoiceIdentity(entities, descriptor, idFactory);
|
|
677
|
+
const existing = linked[0];
|
|
678
|
+
if (existing === void 0) relations.push({
|
|
679
|
+
relationId: createRelationId(idFactory("relation")),
|
|
680
|
+
relationKind: "voice-timbre",
|
|
681
|
+
endpoint0EntityId: createEntityId(audioEntityId),
|
|
682
|
+
endpoint1EntityId: createEntityId(voice.entityId),
|
|
683
|
+
metadata: {},
|
|
684
|
+
trace: {}
|
|
685
|
+
});
|
|
686
|
+
else if (existing.endpoint0EntityId !== audioEntityId || existing.endpoint1EntityId !== voice.entityId) throw new Error(`Voiceover ${audioEntityId} already carries a different Voice timbre`);
|
|
687
|
+
return { voiceEntityId: voice.entityId };
|
|
688
|
+
}
|
|
689
|
+
function appendVoiceIdentity(entities, descriptor, idFactory) {
|
|
690
|
+
const voice = {
|
|
691
|
+
entityId: createEntityId(idFactory("entity")),
|
|
692
|
+
entityKind: "voice",
|
|
693
|
+
payload: { voice: {
|
|
694
|
+
system: descriptor.system,
|
|
695
|
+
key: descriptor.key,
|
|
696
|
+
...descriptor.name === void 0 ? {} : { name: descriptor.name }
|
|
697
|
+
} }
|
|
698
|
+
};
|
|
699
|
+
entities.push(voice);
|
|
700
|
+
return voice;
|
|
701
|
+
}
|
|
702
|
+
/** Two takes share a Voice identity only when they name the same library entry. */
|
|
703
|
+
function sameVoiceDescriptor(stored, descriptor) {
|
|
704
|
+
return isJsonObject(stored) && stored.system === descriptor.system && stored.key === descriptor.key && stored.name === descriptor.name;
|
|
661
705
|
}
|
|
662
706
|
function replaceEntity$1(entities, entityId, replacement) {
|
|
663
707
|
const index = entities.findIndex((row) => row.entityId === entityId);
|
|
664
708
|
if (index < 0) throw new Error(`Entity id ${entityId} does not exist`);
|
|
665
709
|
entities[index] = replacement;
|
|
666
710
|
}
|
|
667
|
-
function sameVoice(left, right) {
|
|
668
|
-
if (right === void 0) return left === void 0;
|
|
669
|
-
return isJsonObject(left) && left.system === right.system && left.key === right.key && left.name === right.name && Object.keys(left).every((key) => key === "system" || key === "key" || key === "name");
|
|
670
|
-
}
|
|
671
711
|
//#endregion
|
|
672
712
|
//#region src/entity-editor/entity-timeline-editor.ts
|
|
673
713
|
/**
|
|
@@ -1203,20 +1243,20 @@ var EntityTimelineEditor = class {
|
|
|
1203
1243
|
hostClipEntityId: input.hostClipEntityId,
|
|
1204
1244
|
anchorOffset: input.anchorOffset
|
|
1205
1245
|
};
|
|
1206
|
-
if (input.placement !== void 0 && (input.hostClipEntityId !== void 0 || input.anchorOffset !== void 0)) throw new Error("Specify
|
|
1246
|
+
if (input.placement !== void 0 && (input.hostClipEntityId !== void 0 || input.anchorOffset !== void 0)) throw new Error("Specify voiceover placement or hostClipEntityId/anchorOffset, not both");
|
|
1207
1247
|
requireVolume(input.volume);
|
|
1208
1248
|
const imported = importMediaAsset(draft, input.media, this.idFactory);
|
|
1209
1249
|
replaceRows(draft, imported.rows);
|
|
1210
|
-
const
|
|
1250
|
+
const voiceover = requireEntityKind(draft, imported.contentEntityId, "audio");
|
|
1211
1251
|
const phonetic = this.requireComposedPhoneticScript(draft, input.phoneticScriptEntityId);
|
|
1212
|
-
this.ensurePhoneticRender(draft,
|
|
1252
|
+
this.ensurePhoneticRender(draft, voiceover.entityId, phonetic.entityId);
|
|
1213
1253
|
const speechTrack = this.ensureRoleTrack(draft, input.timelineEntityId, "speech");
|
|
1214
1254
|
const existing = draft.entities.find((row) => row.entityId === input.voiceoverClipEntityId);
|
|
1215
1255
|
let voiceoverClipEntityId;
|
|
1216
1256
|
if (existing === void 0) voiceoverClipEntityId = this.insertPlacedClipIntoDraft(draft, {
|
|
1217
1257
|
clipEntityId: input.voiceoverClipEntityId,
|
|
1218
1258
|
trackEntityId: speechTrack.entityId,
|
|
1219
|
-
contentEntityId:
|
|
1259
|
+
contentEntityId: voiceover.entityId,
|
|
1220
1260
|
sourceRange: {
|
|
1221
1261
|
start: 0,
|
|
1222
1262
|
end: input.media.durationMs
|
|
@@ -1240,7 +1280,7 @@ var EntityTimelineEditor = class {
|
|
|
1240
1280
|
});
|
|
1241
1281
|
this.relinkRelation(draft, topology.markerContentRelation, {
|
|
1242
1282
|
endpoint0EntityId: topology.marker.entityId,
|
|
1243
|
-
endpoint1EntityId:
|
|
1283
|
+
endpoint1EntityId: voiceover.entityId
|
|
1244
1284
|
});
|
|
1245
1285
|
this.applyClipPlacement(draft, topology.clip.entityId, placement);
|
|
1246
1286
|
const placed = requireClipTopology(draft, topology.clip.entityId);
|
|
@@ -1259,11 +1299,12 @@ var EntityTimelineEditor = class {
|
|
|
1259
1299
|
}
|
|
1260
1300
|
});
|
|
1261
1301
|
}
|
|
1262
|
-
this.reconcileVoiceoverCaptions(draft, voiceoverClipEntityId,
|
|
1302
|
+
this.reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voiceover.entityId, phonetic.entityId, input.captions);
|
|
1263
1303
|
this.resolveOneVoiceoverOverlap(draft);
|
|
1264
1304
|
return {
|
|
1265
1305
|
voiceoverClipEntityId,
|
|
1266
|
-
|
|
1306
|
+
voiceoverAudioEntityId: voiceover.entityId,
|
|
1307
|
+
...imported.voiceEntityId === void 0 ? {} : { voiceEntityId: imported.voiceEntityId },
|
|
1267
1308
|
phoneticScriptEntityId: phonetic.entityId,
|
|
1268
1309
|
audioScriptEntityId: requireComposedAudioScript(draft, phonetic.entityId),
|
|
1269
1310
|
captionClipEntityIds: input.captions.map((caption) => caption.captionClipEntityId)
|
|
@@ -1469,23 +1510,23 @@ var EntityTimelineEditor = class {
|
|
|
1469
1510
|
draft.relations.push(this.createRelation(draft, "timeline-track", timeline.entityId, track.entityId));
|
|
1470
1511
|
return track;
|
|
1471
1512
|
}
|
|
1472
|
-
/** The existing pronunciation variant a
|
|
1513
|
+
/** The existing pronunciation variant a voiceover consumes; composition is required. */
|
|
1473
1514
|
requireComposedPhoneticScript(draft, phoneticScriptEntityId) {
|
|
1474
1515
|
const phonetic = requireEntityKind(draft, phoneticScriptEntityId, "phonetic-script");
|
|
1475
1516
|
requireComposedAudioScript(draft, phonetic.entityId);
|
|
1476
1517
|
return phonetic;
|
|
1477
1518
|
}
|
|
1478
1519
|
/** Preserve the factual synthesis input, regardless of persisted endpoint positions. */
|
|
1479
|
-
ensurePhoneticRender(draft,
|
|
1480
|
-
const renders = draft.relations.filter((relation) => relation.relationKind === "phonetic-script-render" && (relation.endpoint0EntityId ===
|
|
1481
|
-
if (renders.length > 1) throw new Error(`
|
|
1520
|
+
ensurePhoneticRender(draft, voiceoverAudioEntityId, phoneticScriptEntityId) {
|
|
1521
|
+
const renders = draft.relations.filter((relation) => relation.relationKind === "phonetic-script-render" && (relation.endpoint0EntityId === voiceoverAudioEntityId || relation.endpoint1EntityId === voiceoverAudioEntityId));
|
|
1522
|
+
if (renders.length > 1) throw new Error(`Voiceover "${voiceoverAudioEntityId}" has multiple PhoneticScript render relations`);
|
|
1482
1523
|
if (renders[0] !== void 0) {
|
|
1483
|
-
if (otherEndpoint(renders[0],
|
|
1484
|
-
throw new Error(`
|
|
1524
|
+
if (otherEndpoint(renders[0], voiceoverAudioEntityId) === phoneticScriptEntityId) return;
|
|
1525
|
+
throw new Error(`Voiceover "${voiceoverAudioEntityId}" already has a different PhoneticScript synthesis input`);
|
|
1485
1526
|
}
|
|
1486
|
-
draft.relations.push(this.createRelation(draft, "phonetic-script-render",
|
|
1527
|
+
draft.relations.push(this.createRelation(draft, "phonetic-script-render", voiceoverAudioEntityId, phoneticScriptEntityId));
|
|
1487
1528
|
}
|
|
1488
|
-
reconcileVoiceoverCaptions(draft, voiceoverClipEntityId,
|
|
1529
|
+
reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voiceoverAudioEntityId, phoneticScriptEntityId, captions) {
|
|
1489
1530
|
const audioScriptEntityId = requireComposedAudioScript(draft, phoneticScriptEntityId);
|
|
1490
1531
|
const requested = new Set(captions.map((caption) => caption.captionClipEntityId));
|
|
1491
1532
|
if (requested.size !== captions.length) throw new Error("Caption Clip ids must be unique within a voiceover take");
|
|
@@ -1516,7 +1557,7 @@ var EntityTimelineEditor = class {
|
|
|
1516
1557
|
};
|
|
1517
1558
|
draft.entities.push(content);
|
|
1518
1559
|
assertCaptionScript(draft, content.entityId, audioScriptEntityId);
|
|
1519
|
-
draft.relations.push(this.createRelation(draft, "caption-alignment", content.entityId,
|
|
1560
|
+
draft.relations.push(this.createRelation(draft, "caption-alignment", content.entityId, voiceoverAudioEntityId, { alignment: {
|
|
1520
1561
|
start: startMs,
|
|
1521
1562
|
end: startMs + durationMs
|
|
1522
1563
|
} }));
|
|
@@ -1573,7 +1614,7 @@ var EntityTimelineEditor = class {
|
|
|
1573
1614
|
}
|
|
1574
1615
|
}
|
|
1575
1616
|
});
|
|
1576
|
-
this.replaceCaptionAlignment(draft, content.entityId,
|
|
1617
|
+
this.replaceCaptionAlignment(draft, content.entityId, voiceoverAudioEntityId, startMs, durationMs);
|
|
1577
1618
|
}
|
|
1578
1619
|
}
|
|
1579
1620
|
ensureRoleTrackForClip(draft, clipEntityId, role) {
|
|
@@ -1582,10 +1623,10 @@ var EntityTimelineEditor = class {
|
|
|
1582
1623
|
const timeline = requireEntityKind(draft, otherEndpoint(requireOneRelation(draft, track.entityId, "timeline-track"), track.entityId), "timeline");
|
|
1583
1624
|
return this.ensureRoleTrack(draft, timeline.entityId, role);
|
|
1584
1625
|
}
|
|
1585
|
-
replaceCaptionAlignment(draft, captionEntityId,
|
|
1626
|
+
replaceCaptionAlignment(draft, captionEntityId, voiceoverAudioEntityId, startMs, durationMs) {
|
|
1586
1627
|
const removed = new Set(incidentRelations(draft, captionEntityId).filter((relation) => relation.relationKind === "caption-alignment").map((relation) => relation.relationId));
|
|
1587
1628
|
draft.relations = draft.relations.filter((relation) => !removed.has(relation.relationId));
|
|
1588
|
-
draft.relations.push(this.createRelation(draft, "caption-alignment", captionEntityId,
|
|
1629
|
+
draft.relations.push(this.createRelation(draft, "caption-alignment", captionEntityId, voiceoverAudioEntityId, { alignment: {
|
|
1589
1630
|
start: startMs,
|
|
1590
1631
|
end: startMs + durationMs
|
|
1591
1632
|
} }));
|
|
@@ -2048,7 +2089,7 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
|
|
|
2048
2089
|
relations
|
|
2049
2090
|
}, {
|
|
2050
2091
|
assetId: speech.origin_speech_id,
|
|
2051
|
-
kind: "
|
|
2092
|
+
kind: "voiceover",
|
|
2052
2093
|
durationMs: speech.media_duration_ms,
|
|
2053
2094
|
storageKey: speech.audio_storage_key,
|
|
2054
2095
|
voice: {
|
|
@@ -2136,8 +2177,8 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
|
|
|
2136
2177
|
} else if (part.caption !== void 0) {
|
|
2137
2178
|
const caption = part.caption;
|
|
2138
2179
|
duration = caption.initial_duration_ms;
|
|
2139
|
-
const
|
|
2140
|
-
const render = relations.find((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId ===
|
|
2180
|
+
const voiceover = contentByPartId.get(caption.speech_part_id);
|
|
2181
|
+
const render = relations.find((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voiceover);
|
|
2141
2182
|
const phonetic = entities.find((entity) => entity.entityId === render?.endpoint1EntityId);
|
|
2142
2183
|
const script = phonetic === void 0 ? void 0 : composedAudioScript(phonetic, entities);
|
|
2143
2184
|
if (script === void 0) fail("Legacy Caption has no composed AudioScript");
|
|
@@ -2164,7 +2205,7 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
|
|
|
2164
2205
|
selection,
|
|
2165
2206
|
...caption.style === void 0 ? {} : { style: toEntityCaptionStyle(caption.style) }
|
|
2166
2207
|
});
|
|
2167
|
-
link("caption-alignment", content,
|
|
2208
|
+
link("caption-alignment", content, voiceover, { alignment: {
|
|
2168
2209
|
start: caption.start_ms,
|
|
2169
2210
|
end: caption.start_ms + duration
|
|
2170
2211
|
} });
|
|
@@ -2503,21 +2544,21 @@ function requirePart(document, partId) {
|
|
|
2503
2544
|
if (part === void 0) fail(`Placed Clip ${partId} has no part payload`);
|
|
2504
2545
|
return part;
|
|
2505
2546
|
}
|
|
2506
|
-
function ensureSpeechPhoneticChain(
|
|
2507
|
-
const renders = relations.filter((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId ===
|
|
2508
|
-
if (renders.length > 1) fail(`
|
|
2547
|
+
function ensureSpeechPhoneticChain(voiceoverAudioEntityId, audioScriptText, phonemeScript, entities, relations, addEntity, link, documentAudioScriptEntityId) {
|
|
2548
|
+
const renders = relations.filter((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voiceoverAudioEntityId);
|
|
2549
|
+
if (renders.length > 1) fail(`Voiceover ${voiceoverAudioEntityId} has ambiguous PhoneticScript provenance`);
|
|
2509
2550
|
if (renders.length === 1) {
|
|
2510
2551
|
const phonetic = entities.find((entity) => entity.entityId === renders[0].endpoint1EntityId && entity.entityKind === "phonetic-script");
|
|
2511
|
-
if (phonetic === void 0) fail(`
|
|
2552
|
+
if (phonetic === void 0) fail(`Voiceover ${voiceoverAudioEntityId} has a dangling PhoneticScript render`);
|
|
2512
2553
|
const script = composedAudioScript(phonetic, entities);
|
|
2513
|
-
if (script === void 0 || scriptText(script) !== audioScriptText) fail(`
|
|
2554
|
+
if (script === void 0 || scriptText(script) !== audioScriptText) fail(`Voiceover ${voiceoverAudioEntityId} conflicts with the legacy Speech script`);
|
|
2514
2555
|
return phonetic.entityId;
|
|
2515
2556
|
}
|
|
2516
2557
|
const script = entities.find((entity) => entity.entityId === documentAudioScriptEntityId);
|
|
2517
2558
|
if (script === void 0 || script.entityKind !== "audio-script") fail("Migration must carry the fixed document AudioScript");
|
|
2518
2559
|
const segments = script.payload.segments;
|
|
2519
2560
|
if (Array.isArray(segments) && segments.length > 0) {
|
|
2520
|
-
if (scriptText(script) !== audioScriptText) fail(`
|
|
2561
|
+
if (scriptText(script) !== audioScriptText) fail(`Voiceover ${voiceoverAudioEntityId} conflicts with the project AudioScript text`);
|
|
2521
2562
|
} else {
|
|
2522
2563
|
const scriptIndex = entities.findIndex((row) => row.entityId === script.entityId);
|
|
2523
2564
|
entities[scriptIndex] = {
|
|
@@ -2532,7 +2573,7 @@ function ensureSpeechPhoneticChain(voiceEntityId, audioScriptText, phonemeScript
|
|
|
2532
2573
|
baseEntityIds: [script.entityId],
|
|
2533
2574
|
...phonemeScript === void 0 ? {} : { phonemeScript }
|
|
2534
2575
|
});
|
|
2535
|
-
link("phonetic-script-render",
|
|
2576
|
+
link("phonetic-script-render", voiceoverAudioEntityId, phonetic);
|
|
2536
2577
|
return phonetic;
|
|
2537
2578
|
}
|
|
2538
2579
|
function composedAudioScript(phonetic, entities) {
|
package/dist/testing.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { St as bytesToBase64, o as createMirrorVideoDocumentAdapter, xt as base64ToBytes } from "./document-
|
|
1
|
+
import { St as bytesToBase64, o as createMirrorVideoDocumentAdapter, xt as base64ToBytes } from "./document-DdwzfZ1g.js";
|
|
2
2
|
import { t as classifyUpdate } from "./loro-relay-doc-cJSY-uau.js";
|
|
3
3
|
import { LoroDoc, VersionVector, encodeFrontiers } from "loro-crdt";
|
|
4
4
|
//#region src/testing/in-memory-mengine-server.ts
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@mengine/medeo-client",
|
|
3
|
-
"version": "2.0.1-alpha.
|
|
3
|
+
"version": "2.0.1-alpha.8",
|
|
4
4
|
"license": "UNLICENSED",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -28,9 +28,9 @@
|
|
|
28
28
|
"immer": "^10.2.0",
|
|
29
29
|
"loro-mirror": "^2.3.3",
|
|
30
30
|
"zod": "^4.4.3",
|
|
31
|
-
"@mengine/
|
|
32
|
-
"@mengine/
|
|
33
|
-
"@mengine/
|
|
31
|
+
"@mengine/sync": "2.0.1-alpha.8",
|
|
32
|
+
"@mengine/utils": "2.0.1-alpha.8",
|
|
33
|
+
"@mengine/storage": "2.0.1-alpha.8"
|
|
34
34
|
},
|
|
35
35
|
"devDependencies": {
|
|
36
36
|
"@types/node": "^25.9.1",
|