@mengine/medeo-client 2.0.1-alpha.7 → 2.0.1-alpha.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1447,8 +1447,85 @@ function emptyLaneTracks() {
1447
1447
  }
1448
1448
  //#endregion
1449
1449
  //#region ../medeo-dsl/src/entities.ts
1450
+ /**
1451
+ * What every known entity kind declares.
1452
+ *
1453
+ * `kinds` is a multi-declaration: every entry holds at the same time, the first
1454
+ * naming the object and the rest the properties it shares with others. The
1455
+ * order carries no priority, endpoint role, or edit sequence — an operation
1456
+ * asks whether the entity declares the kind it needs, never what its "main
1457
+ * type" is.
1458
+ *
1459
+ * `Caption` declares `Sequence` on purpose: it owns the display timing of the
1460
+ * one AudioScript segment it shows, which is what admits it into a Clip.
1461
+ * `Voice` declares neither `MediaFile` nor `Sequence` — it is a timbre
1462
+ * identity, not playable content; a rendered voiceover is an `audio` entity.
1463
+ */
1464
+ const ENTITY_KINDS = Object.freeze({
1465
+ axvideo: Object.freeze([
1466
+ "AXVideo",
1467
+ "Container",
1468
+ "Sequence",
1469
+ "Visual",
1470
+ "Audible"
1471
+ ]),
1472
+ timeline: Object.freeze(["Timeline", "Container"]),
1473
+ track: Object.freeze([
1474
+ "Track",
1475
+ "Container",
1476
+ "Sequence"
1477
+ ]),
1478
+ clip: Object.freeze(["Clip", "Container"]),
1479
+ asset: Object.freeze(["Asset"]),
1480
+ video: Object.freeze([
1481
+ "Video",
1482
+ "MediaFile",
1483
+ "Sequence",
1484
+ "Visual",
1485
+ "Audible"
1486
+ ]),
1487
+ audio: Object.freeze([
1488
+ "Audio",
1489
+ "MediaFile",
1490
+ "Sequence",
1491
+ "Audible"
1492
+ ]),
1493
+ image: Object.freeze([
1494
+ "Image",
1495
+ "MediaFile",
1496
+ "Sequence",
1497
+ "Visual"
1498
+ ]),
1499
+ voice: Object.freeze([
1500
+ "Voice",
1501
+ "Container",
1502
+ "IPBound"
1503
+ ]),
1504
+ "sequence-marker": Object.freeze(["SequenceMarker"]),
1505
+ viewport: Object.freeze(["Viewport"]),
1506
+ "audio-script": Object.freeze(["AudioScript", "SegmentText"]),
1507
+ "phonetic-script": Object.freeze(["PhoneticScript", "SegmentText"]),
1508
+ caption: Object.freeze([
1509
+ "Caption",
1510
+ "SegmentText",
1511
+ "Sequence"
1512
+ ])
1513
+ });
1514
+ /** What a known kind declares; extension kinds declare nothing through this table. */
1515
+ function kindsOf(entityKind) {
1516
+ return isKnownEntityKind(entityKind) ? ENTITY_KINDS[entityKind] : [];
1517
+ }
1518
+ function declaresKind(entityKind, kind) {
1519
+ return kindsOf(entityKind).includes(kind);
1520
+ }
1521
+ /**
1522
+ * Sequence admission asks `kinds` first, so an entity that does not declare
1523
+ * `Sequence` can never buy its way in by carrying the fields. Extension kinds
1524
+ * declare nothing and stay structurally detected.
1525
+ */
1450
1526
  function hasSequence(entity) {
1451
- if (isKnownNonSequenceKind(entity.entityKind) || isReservedEntityKind(entity.entityKind)) return false;
1527
+ if (isReservedEntityKind(entity.entityKind)) return false;
1528
+ if (isKnownEntityKind(entity.entityKind) && !declaresKind(entity.entityKind, "Sequence")) return false;
1452
1529
  return isSequenceFields(entity);
1453
1530
  }
1454
1531
  function isSequenceFields(value) {
@@ -1461,22 +1538,16 @@ function isSequenceFields(value) {
1461
1538
  if (value.sampling !== "native" && value.sampling !== "constant" && value.sampling !== "derived") return false;
1462
1539
  return "coordinateSpace" in value && value.coordinateSpace !== void 0;
1463
1540
  }
1464
- function isKnownSequenceKind(kind) {
1465
- return isMediaAssetVariantKind(kind) || kind === "caption" || kind === "axvideo";
1466
- }
1467
1541
  function isMediaAssetVariantKind(kind) {
1468
- return kind === "video" || kind === "image" || kind === "audio" || kind === "voice";
1542
+ return kind === "video" || kind === "image" || kind === "audio";
1469
1543
  }
1470
1544
  function isKnownEntityKind(kind) {
1471
- return isKnownSequenceKind(kind) || isKnownNonSequenceKind(kind);
1545
+ return Object.hasOwn(ENTITY_KINDS, kind);
1472
1546
  }
1473
1547
  function isReservedEntityKind(kind) {
1474
1548
  const normalized = kind.toLowerCase().replaceAll("-", "").replaceAll("_", "");
1475
1549
  return normalized === "speech" || normalized === "videodocument";
1476
1550
  }
1477
- function isKnownNonSequenceKind(kind) {
1478
- return kind === "timeline" || kind === "track" || kind === "clip" || kind === "asset" || kind === "sequence-marker" || kind === "viewport" || kind === "audio-script" || kind === "phonetic-script";
1479
- }
1480
1551
  function isRecord$1(value) {
1481
1552
  return typeof value === "object" && value != null && !Array.isArray(value);
1482
1553
  }
@@ -1660,14 +1731,17 @@ function assembleCaptionContent(rows, captionEntityId) {
1660
1731
  requireEntityKind(rows, captionEntityId, "caption");
1661
1732
  const caption = assembleEntityContent(rows, captionEntityId);
1662
1733
  const script = findComposedAudioScript(rows, captionEntityId, "caption");
1663
- const segment = selectAudioScriptSegment({
1734
+ const selection = captionSelection(caption);
1735
+ const composed = {
1664
1736
  ...script,
1665
1737
  payload: caption.payload
1666
- }, captionSelection(caption));
1738
+ };
1739
+ const segment = selectAudioScriptSegment(composed, selection);
1667
1740
  return {
1668
1741
  caption,
1669
1742
  audioScript: script,
1670
1743
  segments: [segment],
1744
+ segmentIndex: scriptSegments(composed).findIndex((entry) => entry.segmentId === selection.segmentId),
1671
1745
  text: segment.text
1672
1746
  };
1673
1747
  }
@@ -2059,12 +2133,15 @@ function validateKnownEntityPayload(entity) {
2059
2133
  break;
2060
2134
  case "video":
2061
2135
  case "audio":
2062
- case "voice":
2136
+ validateSequencePayload(value, "bounded", "native", problems);
2137
+ validateMediaAssetFields(value, problems);
2138
+ break;
2063
2139
  case "caption":
2064
2140
  validateSequencePayload(value, "bounded", "native", problems);
2065
- if (entity.entityKind !== "caption") validateMediaAssetFields(value, problems);
2066
- if (entity.entityKind === "voice") validateVoicePayload(value, problems);
2067
- if (entity.entityKind === "caption") validateCaptionPayload(value, problems);
2141
+ validateCaptionPayload(value, problems);
2142
+ break;
2143
+ case "voice":
2144
+ validateVoicePayload(value, problems);
2068
2145
  break;
2069
2146
  case "image":
2070
2147
  validateSequencePayload(value, "unbounded", "constant", problems);
@@ -2171,11 +2248,26 @@ function validateExternalLocator(value, problems) {
2171
2248
  function validateStorageKey(value, problems) {
2172
2249
  if (value.storageKey !== void 0 && (typeof value.storageKey !== "string" || value.storageKey.trim() === "")) problems.push("storageKey must be a non-empty string when present");
2173
2250
  }
2251
+ /**
2252
+ * Voice is a timbre identity, not playable content: it owns the voice-library
2253
+ * descriptor and nothing that would make it look like media or a Sequence. The
2254
+ * rendered voiceover is an `audio` entity bound by a `voice-timbre` Relation.
2255
+ */
2174
2256
  function validateVoicePayload(value, problems) {
2257
+ for (const key of [
2258
+ "extent",
2259
+ "sampling",
2260
+ "coordinateSpace",
2261
+ "external",
2262
+ "storageKey"
2263
+ ]) if (Object.hasOwn(value, key)) problems.push(`${key} belongs to the rendered Audio, not to the Voice identity`);
2175
2264
  const voice = value.voice;
2176
- if (voice === void 0) return;
2265
+ if (voice === void 0) {
2266
+ problems.push("voice is required; a Voice entity is its voice-library identity");
2267
+ return;
2268
+ }
2177
2269
  if (!isRecord(voice)) {
2178
- problems.push("voice must be an object when present");
2270
+ problems.push("voice must be an object");
2179
2271
  return;
2180
2272
  }
2181
2273
  if (voice.system !== "voice-library") problems.push("voice.system must be \"voice-library\"");
@@ -2307,7 +2399,6 @@ function isOwnedLocalEntityValuePath(entityKind, path) {
2307
2399
  if ([
2308
2400
  "video",
2309
2401
  "audio",
2310
- "voice",
2311
2402
  "image",
2312
2403
  "caption",
2313
2404
  "axvideo"
@@ -2371,7 +2462,7 @@ function collectRelatedEntities(entityRefs, index) {
2371
2462
  };
2372
2463
  }
2373
2464
  function expectedSequenceShape(kind) {
2374
- if (kind === "video" || kind === "audio" || kind === "voice" || kind === "caption") return {
2465
+ if (kind === "video" || kind === "audio" || kind === "caption") return {
2375
2466
  extent: "bounded",
2376
2467
  sampling: "native"
2377
2468
  };
@@ -2412,7 +2503,7 @@ const generatedRelationSpec = Object.freeze({
2412
2503
  });
2413
2504
  const captionAlignmentRelationSpec = Object.freeze({
2414
2505
  kind: "caption-alignment",
2415
- validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["caption"]), new Set(["audio", "voice"])),
2506
+ validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["caption"]), new Set(["audio"])),
2416
2507
  validateMetadata: isCaptionAlignmentMetadata
2417
2508
  });
2418
2509
  /** `clip-anchor(child, host)` means endpoint 0 follows endpoint 1. */
@@ -2421,20 +2512,29 @@ const clipAnchorRelationSpec = Object.freeze({
2421
2512
  validateEndpoints: (endpoints) => endpoints[0].current().entityKind === "clip" && endpoints[1].current().entityKind === "clip",
2422
2513
  validateMetadata: isEmptyMetadata
2423
2514
  });
2424
- /** Voice is generated from PhoneticScript; kinds determine roles regardless of endpoint positions. */
2515
+ /**
2516
+ * The rendered voiceover Audio and the PhoneticScript it was synthesized from;
2517
+ * kinds determine roles regardless of endpoint positions.
2518
+ */
2425
2519
  const phoneticScriptRenderRelationSpec = Object.freeze({
2426
2520
  kind: "phonetic-script-render",
2427
- validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["voice"]), new Set(["phonetic-script"])),
2521
+ validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio"]), new Set(["phonetic-script"])),
2522
+ validateMetadata: isEmptyMetadata
2523
+ });
2524
+ /**
2525
+ * The timbre identity a rendered voiceover Audio was synthesized with. Voice
2526
+ * stays an identity: it never carries the audio itself, so the link between the
2527
+ * two is a Relation rather than one shared entity.
2528
+ */
2529
+ const voiceTimbreRelationSpec = Object.freeze({
2530
+ kind: "voice-timbre",
2531
+ validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio"]), new Set(["voice"])),
2428
2532
  validateMetadata: isEmptyMetadata
2429
2533
  });
2430
2534
  /** AudioScript was transcribed from Audio, Video, or recorded Voice; kinds determine roles. */
2431
2535
  const audioScriptSourceRelationSpec = Object.freeze({
2432
2536
  kind: "audio-script-source",
2433
- validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio-script"]), new Set([
2434
- "audio",
2435
- "video",
2436
- "voice"
2437
- ])),
2537
+ validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio-script"]), new Set(["audio", "video"])),
2438
2538
  validateMetadata: isEmptyMetadata
2439
2539
  });
2440
2540
  /**
@@ -2461,6 +2561,7 @@ const builtInRelationSpecs = Object.freeze([
2461
2561
  captionAlignmentRelationSpec,
2462
2562
  clipAnchorRelationSpec,
2463
2563
  phoneticScriptRenderRelationSpec,
2564
+ voiceTimbreRelationSpec,
2464
2565
  audioScriptSourceRelationSpec,
2465
2566
  audioScriptMarkerRelationSpec
2466
2567
  ]);
@@ -2490,7 +2591,7 @@ function hasAssetAndSequence(endpoints) {
2490
2591
  return first.entityKind === "asset" && hasSequence(second) || second.entityKind === "asset" && hasSequence(first);
2491
2592
  }
2492
2593
  function isGeneratedMedia(entity) {
2493
- return entity.entityKind === "video" || entity.entityKind === "image" || entity.entityKind === "audio" || entity.entityKind === "voice";
2594
+ return entity.entityKind === "video" || entity.entityKind === "image" || entity.entityKind === "audio";
2494
2595
  }
2495
2596
  function isEmptyMetadata(value) {
2496
2597
  return isJsonObject(value) && Object.keys(value).length === 0;
@@ -2920,7 +3021,7 @@ function resolveGraphLayout(graph) {
2920
3021
  if (clip.entityKind !== "clip" || graph.adjacent(clip.entityId, "track-clip").length !== 1) fail(`Clip ${clip.entityId} must belong to exactly one Track`);
2921
3022
  const marker = graph.one(clip.entityId, "clip-marker", "sequence-marker");
2922
3023
  const content = graph.one(marker.entityId, "marker-content");
2923
- if (!(typedRole === "video_clip" ? ["image", "video"] : typedRole === "speech" ? ["voice"] : typedRole === "caption" ? ["caption"] : ["audio"]).includes(content.entityKind)) fail(`unsupported compatibility content kind ${content.entityKind} on ${typedRole} Track`);
3024
+ if (!(typedRole === "video_clip" ? ["image", "video"] : typedRole === "speech" ? ["audio"] : typedRole === "caption" ? ["caption"] : ["audio"]).includes(content.entityKind)) fail(`unsupported compatibility content kind ${content.entityKind} on ${typedRole} Track`);
2924
3025
  const space = content.payload.coordinateSpace;
2925
3026
  if (space !== "ms" && space !== "milliseconds" && (!object(space) || space.unit !== "ms" && space.unit !== "milliseconds")) fail(`media ${content.entityId} must state millisecond coordinates for this reader`);
2926
3027
  const source = range(marker.payload.sourceRange, "sourceRange");
@@ -2999,7 +3100,7 @@ function resolveGraphLayout(graph) {
2999
3100
  let start = fact.absoluteStart ?? 0;
3000
3101
  if (fact.mode === "anchored") {
3001
3102
  const host = resolve(fact.anchorId);
3002
- if (fact.role === "speech" && !["video_clip", "bgm"].includes(facts.get(host.clipEntityId).role)) fail("An anchored Voice must use a visual or BGM Clip as its host");
3103
+ if (fact.role === "speech" && !["video_clip", "bgm"].includes(facts.get(host.clipEntityId).role)) fail("An anchored voiceover must use a visual or BGM Clip as its host");
3003
3104
  start = host.start + fact.anchorOffset;
3004
3105
  } else if (fact.predecessor !== void 0) {
3005
3106
  const prior = resolve(fact.predecessor);
@@ -3083,8 +3184,11 @@ function projectEntityTimeline(rows) {
3083
3184
  const asset = mediaAssetFacts(graph, content);
3084
3185
  requireUntrimmedAudio(content, marker, placement);
3085
3186
  const external = externalLocator(asset, "memota-speech");
3086
- const assembled = graph.adjacent(content.entityId, "phonetic-script-render").length > 0 || content.payload.voice !== void 0 ? assemblePhoneticScriptContent(rows, graph.one(content.entityId, "phonetic-script-render", "phonetic-script").entityId) : void 0;
3087
- const voice = content.payload.voice;
3187
+ const renderEdges = graph.adjacent(content.entityId, "phonetic-script-render");
3188
+ const timbreEdges = graph.adjacent(content.entityId, "voice-timbre");
3189
+ const assembled = renderEdges.length > 0 || timbreEdges.length > 0 ? assemblePhoneticScriptContent(rows, graph.one(content.entityId, "phonetic-script-render", "phonetic-script").entityId) : void 0;
3190
+ if (timbreEdges.length > 1) fail("A voiceover resolves at most one Voice identity");
3191
+ const voice = timbreEdges.length === 0 ? void 0 : graph.one(content.entityId, "voice-timbre", "voice").payload.voice;
3088
3192
  if (voice !== void 0 && (!object(voice) || voice.system !== "voice-library" || typeof voice.key !== "string")) fail("Voice must use an explicit voice-library locator");
3089
3193
  const captionIds = layout.tracks.flatMap((candidate) => candidate.role === "caption" ? candidate.clips.filter((caption) => caption.anchorClipEntityId === clip.entityId).map((caption) => caption.clipEntityId) : []);
3090
3194
  parts[clip.entityId] = { speech: {
package/dist/index.d.ts CHANGED
@@ -192,6 +192,16 @@ interface AssembledCaptionContent {
192
192
  readonly audioScript: EntityRow<'audio-script'>;
193
193
  /** Selected AudioScript segments in Caption selection order. */
194
194
  readonly segments: readonly ScriptTextSegment[];
195
+ /**
196
+ * Where the selected segment sits in the AudioScript's ordered segments.
197
+ *
198
+ * A Caption locates its text as AudioScript plus index. The selection is
199
+ * stored as the segment's own stable id rather than this ordinal, so a
200
+ * concurrent insert or re-segmentation cannot silently slide a Caption onto
201
+ * different text; the ordinal is derived here for readers that want the
202
+ * position.
203
+ */
204
+ readonly segmentIndex: number;
195
205
  /** Complete display text: the selected segments' text joined in order. */
196
206
  readonly text: string;
197
207
  }
@@ -272,17 +282,22 @@ interface AudioMediaAssetFact {
272
282
  readonly durationMs: number;
273
283
  readonly storageKey: string;
274
284
  }
275
- interface VoiceMediaAssetFact {
285
+ /**
286
+ * A speech recording. It materializes as an `audio` entity like every other
287
+ * playable sound; `voice` names the timbre identity it was synthesized with,
288
+ * which becomes its own Voice entity linked by a `voice-timbre` Relation.
289
+ */
290
+ interface VoiceoverMediaAssetFact {
276
291
  /** Stable external speech result id, independent of the placed Clip id. */
277
292
  readonly assetId: string;
278
- readonly kind: 'voice';
293
+ readonly kind: 'voiceover';
279
294
  readonly durationMs: number;
280
295
  readonly storageKey: string;
281
- /** Present for synthesized voice, absent for original recorded audio. */
296
+ /** Present for synthesized speech, absent for original recorded audio. */
282
297
  readonly voice?: VoiceDescriptor;
283
298
  }
284
299
  /** Facts resolved from media storage. A trim window never substitutes for intrinsic duration. */
285
- type MediaAssetFact = ImageMediaAssetFact | VideoMediaAssetFact | AudioMediaAssetFact | VoiceMediaAssetFact;
300
+ type MediaAssetFact = ImageMediaAssetFact | VideoMediaAssetFact | AudioMediaAssetFact | VoiceoverMediaAssetFact;
286
301
  type ClipPlacement = {
287
302
  readonly kind: 'sequential';
288
303
  readonly order: number;
@@ -334,7 +349,7 @@ interface DeleteClipTreeInput {
334
349
  interface VoiceoverCaptionFact {
335
350
  /** Stable placed caption identity supplied by the materialized side effect. */
336
351
  readonly captionClipEntityId: string;
337
- /** Directly held bases; includes the AudioScript used by the Voice. */
352
+ /** Directly held bases; includes the AudioScript the voiceover was rendered from. */
338
353
  readonly baseEntityIds: readonly string[];
339
354
  /** Ordered selection of AudioScript segments; caption text is never passed inline. */
340
355
  readonly selection: CaptionSegmentSelection;
@@ -345,7 +360,7 @@ interface VoiceoverCaptionFact {
345
360
  type VoiceoverTakeInput = {
346
361
  readonly timelineEntityId: string; /** Stable placed speech identity, distinct from media.assetId. */
347
362
  readonly voiceoverClipEntityId: string;
348
- readonly media: VoiceMediaAssetFact; /** Existing pronunciation variant; its composed AudioScript stays the text owner. */
363
+ readonly media: VoiceoverMediaAssetFact; /** Existing pronunciation variant; its composed AudioScript stays the text owner. */
349
364
  readonly phoneticScriptEntityId: string;
350
365
  readonly volume: number;
351
366
  readonly captions: readonly VoiceoverCaptionFact[];
@@ -421,8 +436,11 @@ interface TrimClipInput {
421
436
  }
422
437
  interface VoiceoverTakeResult {
423
438
  readonly voiceoverClipEntityId: string;
424
- readonly voiceEntityId: string;
425
- /** The pronunciation variant the Voice was rendered from. */
439
+ /** The rendered voiceover Audio entity. */
440
+ readonly voiceoverAudioEntityId: string;
441
+ /** The timbre identity it was synthesized with, when the take declared one. */
442
+ readonly voiceEntityId?: string;
443
+ /** The pronunciation variant the voiceover was rendered from. */
426
444
  readonly phoneticScriptEntityId: string;
427
445
  /** The base-text owner resolved from the PhoneticScript baseEntityIds. */
428
446
  readonly audioScriptEntityId: string;
@@ -556,7 +574,7 @@ declare class EntityTimelineEditor {
556
574
  insertCaptionClip(input: InsertCaptionClipInput): ClipEntityId;
557
575
  private insertPlacedClipIntoDraft;
558
576
  private ensureRoleTrack;
559
- /** The existing pronunciation variant a Voice consumes; composition is required. */
577
+ /** The existing pronunciation variant a voiceover consumes; composition is required. */
560
578
  private requireComposedPhoneticScript;
561
579
  /** Preserve the factual synthesis input, regardless of persisted endpoint positions. */
562
580
  private ensurePhoneticRender;
@@ -632,6 +650,8 @@ interface ImportedMediaAsset {
632
650
  readonly rows: EntityRelationRows;
633
651
  /** The single media variant identity for this asset id; it is its own Asset. */
634
652
  readonly contentEntityId: string;
653
+ /** The timbre identity of a synthesized voiceover; absent for recorded media. */
654
+ readonly voiceEntityId?: string;
635
655
  }
636
656
  /**
637
657
  * Import at the editor boundary, not at asset generation time. Existing
@@ -1716,4 +1736,4 @@ declare function hasAudioScriptAsset(payload: JsonObject): boolean;
1716
1736
  /** Internal fingerprint prevents a resource from silently describing an older text revision. */
1717
1737
  declare function audioScriptAssetFields(payload: JsonObject, assetId: string): JsonObject;
1718
1738
  //#endregion
1719
- export { type Aggregation, type AssembledCaptionContent, type AssembledPhoneticScriptContent, type Attachment, type AudioMediaAssetFact, type BgmPart, type CaptionContentInput, type CaptionDisplayCue, type CaptionPart, type CaptionSegmentSelection, type CaptionStyle, type ClipEntityId, type ClipPlacement, type CommitOptions, DEFAULT_UNIT_TIME_MS, type DeleteBgmInput, type DeleteClipInput, type DeleteClipTreeInput, type DeleteVoiceoverInput, type DerivedItemPosition, type DocStorageLike, type DocVersionMark, type DocumentAudioScript, type DocumentAudioScriptState, type EditorFoundation, EntityDocumentState, type EntityGraphCommitOptions, EntityGraphHttpClient, type EntityGraphState, EntityTimelineEditor, type EntityTimelineIdFactory, EntityTimelineProjectionError, type EntityTimelineTrackRole, type FieldChange, type FieldPath, IMPLEMENTED_SEMANTIC_OP_KINDS, type ImageMediaAssetFact, type ImplementedSemanticOpKind, type ImportedMediaAsset, type InsertCaptionClipInput, type InsertClipInput, type InsertMediaClipInput, type InsertMediaClipsInput, type InsertPlacedClipInput, type JournalEntry, LANE_KINDS_IN_STACK_ORDER, LORO_ENTITY_SCHEMA, type LaneKind, type LinearClipSpeed, LoroEntityDocument, LoroEntityDraft, type MainClipRange, type MakeEmptyPart, ManualSyncDoc, type ManualSyncDocOptions, MedeoHttpDocStorage, type MedeoHttpDocStorageOptions, type MediaAssetFact, type MediaClipInsertion, MemoryDocStorage, MengineAckFailedError, type MengineAuditEntry, type MengineAuditResponse, MengineDocSession, type MengineDocSessionOptions, type MengineDocSessionUpdateEvent, type MengineDocSyncState, type MengineDocumentVersion, type MengineEventStreamOptions, MengineHttpClient, type MengineHttpClientOptions, MengineHttpRequestError, MenginePushRejectedError, type MenginePushResponse, type MenginePushUpdateResponse, type MengineRejectedResponse, type MengineSnapshotResponse, type MengineSseUpdateEvent, type MengineSyncResponse, type MengineUndoState, type MengineUpdateMeta, MirrorVideoDocumentAdapter, type MirrorVideoDocumentOptions, type MoveClipInput, type MoveClipsToStartsInput, type MoveSequentialClipsInput, type MoveVoiceoverInput, type OpActor, type PartAggregation, type PartIdFactory, type PartKind, type PartUnion, type PatchCaptionStyleInput, PlainMemoryAdapter, type PlainMemoryAdapterOptions, type PlannedSemanticOpKind, type PullFailureReason, type PullResult, type PushOutcome, type PushOutcomeKind, type PushResult, type PushResultKind, type ReplaceClipContentInput, type ReplaceMediaClipInput, type ReplaceSequentialClipsInput, type ReplacementMediaClipInput, type ResolvedEntityClipPlacement, type ResolvedEntityTimelineLayout, type ResolvedEntityTrack, SchemaValidator, ScriptCompositionError, type SemanticDocumentAdapter, SemanticEditor, type SemanticOpInput, type SemanticOpKind, type SemanticOpName, type SequentialClipAnchor, type SetBgmInput, type SetCaptionVisibilityInput, type SetClipPlacementInput, type SetClipSpeedInput, type SetClipVolumeInput, type SnapshotReadable, type SolvedVideoDocument, type SpeechHostMap, type SpeechPart, type SpeedShift, TIMELINE_SKELETON_DURATION_MS, type Timeline, type TimelineDoc, type TimelineItem, type Track, type TrackDraft, type TrackItem, type TrackItemDraft, type TrackItemTimePosition, type TransactAudit, type TrimClipInput, type UpdateClipInput, type UpdateClipMarkerInput, ValidationError, type VideoClipPart, type VideoDocument, type VideoDocumentDraft, type VideoDocumentMirrorSchema, VideoDocumentValidationError, type VideoDocumentValidationIssue, type VideoDocumentValidationIssueCode, type VideoDraft, type CaptionPart$1 as VideoDraftCaptionPart, type VideoDraftContent, type VideoDraftPartUnion, type Timeline$1 as VideoDraftTimeline, type Track$1 as VideoDraftTrack, type TrackItem$1 as VideoDraftTrackItem, type VideoMediaAssetFact, type VisualMediaAssetFact, type VoiceMediaAssetFact, type VoiceoverCaptionFact, type VoiceoverTakeInput, type VoiceoverTakeResult, type WaitForServerAckOptions, applyEntityRows, applyFieldChanges, arrangeMainTrackSeamlessly, assembleCaptionContent, assembleEntityContent, assemblePhoneticScriptContent, assembleScriptText, assertCanonicalEditorResources, assertEntityProjectionPreservesLegacy, assertFieldChanges, assertMediaAssetWritePolicy, assertValidVideoDocument, audioScriptAssetContent, audioScriptAssetFields, audioScriptAssetHash, base64ToBytes, buildInitialVideoDocument, buildSpeechHostMap, bytesToBase64, cascadeAfterVideoClipChanges, compileEntityRows, createEditSandbox, createMirrorVideoDocument, createMirrorVideoDocumentAdapter, createPlainMemoryAdapter, decodeDocVersionMark, derivePositionFromAbs, effectiveVideoClipDurationMs, encodeDocVersionMark, ensureEditorFoundation, ensureLaneTrack, fillMainTrackTimeGaps, findComposedAudioScript, findLaneTrack, fromVideoDocument, generatePartId, getAt, hasAudioScriptAsset, hostForAbsMs, importMediaAsset, isEmptyVideoClip, isImplementedSemanticOpKind, isMap, laneTrackId, mainTrackRanges, migrateLegacyTimelineToEntities, partDurationMs, partUnionSchema, prepareCaptionContent, projectEntityTimeline, readDocumentAudioScript, readEntityDocument, readMainTrackItems, readMengineEventStream, readPart, readPartDurationMs, readVideoDocumentFromDraft, reassignSpeechesToVideoClipsByTime, recalculateTimelineDuration, relativePositionForAbs, replayJournal, resolveAllSpeechOverlaps, resolveEntityTimelineLayout, resolveFieldPath, resolveSpeechOverlapByShiftingVideos, safeDurationMs, index_d_exports as schemas, selectAudioScriptSegment, snapshotToPlain, solveVideoDocument, speedOf, syncAggregatedClipsTimePosition, toVideoDocument, validateVideoDocument, videoDocumentMirrorSchema, videoDocumentSchema };
1739
+ export { type Aggregation, type AssembledCaptionContent, type AssembledPhoneticScriptContent, type Attachment, type AudioMediaAssetFact, type BgmPart, type CaptionContentInput, type CaptionDisplayCue, type CaptionPart, type CaptionSegmentSelection, type CaptionStyle, type ClipEntityId, type ClipPlacement, type CommitOptions, DEFAULT_UNIT_TIME_MS, type DeleteBgmInput, type DeleteClipInput, type DeleteClipTreeInput, type DeleteVoiceoverInput, type DerivedItemPosition, type DocStorageLike, type DocVersionMark, type DocumentAudioScript, type DocumentAudioScriptState, type EditorFoundation, EntityDocumentState, type EntityGraphCommitOptions, EntityGraphHttpClient, type EntityGraphState, EntityTimelineEditor, type EntityTimelineIdFactory, EntityTimelineProjectionError, type EntityTimelineTrackRole, type FieldChange, type FieldPath, IMPLEMENTED_SEMANTIC_OP_KINDS, type ImageMediaAssetFact, type ImplementedSemanticOpKind, type ImportedMediaAsset, type InsertCaptionClipInput, type InsertClipInput, type InsertMediaClipInput, type InsertMediaClipsInput, type InsertPlacedClipInput, type JournalEntry, LANE_KINDS_IN_STACK_ORDER, LORO_ENTITY_SCHEMA, type LaneKind, type LinearClipSpeed, LoroEntityDocument, LoroEntityDraft, type MainClipRange, type MakeEmptyPart, ManualSyncDoc, type ManualSyncDocOptions, MedeoHttpDocStorage, type MedeoHttpDocStorageOptions, type MediaAssetFact, type MediaClipInsertion, MemoryDocStorage, MengineAckFailedError, type MengineAuditEntry, type MengineAuditResponse, MengineDocSession, type MengineDocSessionOptions, type MengineDocSessionUpdateEvent, type MengineDocSyncState, type MengineDocumentVersion, type MengineEventStreamOptions, MengineHttpClient, type MengineHttpClientOptions, MengineHttpRequestError, MenginePushRejectedError, type MenginePushResponse, type MenginePushUpdateResponse, type MengineRejectedResponse, type MengineSnapshotResponse, type MengineSseUpdateEvent, type MengineSyncResponse, type MengineUndoState, type MengineUpdateMeta, MirrorVideoDocumentAdapter, type MirrorVideoDocumentOptions, type MoveClipInput, type MoveClipsToStartsInput, type MoveSequentialClipsInput, type MoveVoiceoverInput, type OpActor, type PartAggregation, type PartIdFactory, type PartKind, type PartUnion, type PatchCaptionStyleInput, PlainMemoryAdapter, type PlainMemoryAdapterOptions, type PlannedSemanticOpKind, type PullFailureReason, type PullResult, type PushOutcome, type PushOutcomeKind, type PushResult, type PushResultKind, type ReplaceClipContentInput, type ReplaceMediaClipInput, type ReplaceSequentialClipsInput, type ReplacementMediaClipInput, type ResolvedEntityClipPlacement, type ResolvedEntityTimelineLayout, type ResolvedEntityTrack, SchemaValidator, ScriptCompositionError, type SemanticDocumentAdapter, SemanticEditor, type SemanticOpInput, type SemanticOpKind, type SemanticOpName, type SequentialClipAnchor, type SetBgmInput, type SetCaptionVisibilityInput, type SetClipPlacementInput, type SetClipSpeedInput, type SetClipVolumeInput, type SnapshotReadable, type SolvedVideoDocument, type SpeechHostMap, type SpeechPart, type SpeedShift, TIMELINE_SKELETON_DURATION_MS, type Timeline, type TimelineDoc, type TimelineItem, type Track, type TrackDraft, type TrackItem, type TrackItemDraft, type TrackItemTimePosition, type TransactAudit, type TrimClipInput, type UpdateClipInput, type UpdateClipMarkerInput, ValidationError, type VideoClipPart, type VideoDocument, type VideoDocumentDraft, type VideoDocumentMirrorSchema, VideoDocumentValidationError, type VideoDocumentValidationIssue, type VideoDocumentValidationIssueCode, type VideoDraft, type CaptionPart$1 as VideoDraftCaptionPart, type VideoDraftContent, type VideoDraftPartUnion, type Timeline$1 as VideoDraftTimeline, type Track$1 as VideoDraftTrack, type TrackItem$1 as VideoDraftTrackItem, type VideoMediaAssetFact, type VisualMediaAssetFact, type VoiceoverCaptionFact, type VoiceoverMediaAssetFact, type VoiceoverTakeInput, type VoiceoverTakeResult, type WaitForServerAckOptions, applyEntityRows, applyFieldChanges, arrangeMainTrackSeamlessly, assembleCaptionContent, assembleEntityContent, assemblePhoneticScriptContent, assembleScriptText, assertCanonicalEditorResources, assertEntityProjectionPreservesLegacy, assertFieldChanges, assertMediaAssetWritePolicy, assertValidVideoDocument, audioScriptAssetContent, audioScriptAssetFields, audioScriptAssetHash, base64ToBytes, buildInitialVideoDocument, buildSpeechHostMap, bytesToBase64, cascadeAfterVideoClipChanges, compileEntityRows, createEditSandbox, createMirrorVideoDocument, createMirrorVideoDocumentAdapter, createPlainMemoryAdapter, decodeDocVersionMark, derivePositionFromAbs, effectiveVideoClipDurationMs, encodeDocVersionMark, ensureEditorFoundation, ensureLaneTrack, fillMainTrackTimeGaps, findComposedAudioScript, findLaneTrack, fromVideoDocument, generatePartId, getAt, hasAudioScriptAsset, hostForAbsMs, importMediaAsset, isEmptyVideoClip, isImplementedSemanticOpKind, isMap, laneTrackId, mainTrackRanges, migrateLegacyTimelineToEntities, partDurationMs, partUnionSchema, prepareCaptionContent, projectEntityTimeline, readDocumentAudioScript, readEntityDocument, readMainTrackItems, readMengineEventStream, readPart, readPartDurationMs, readVideoDocumentFromDraft, reassignSpeechesToVideoClipsByTime, recalculateTimelineDuration, relativePositionForAbs, replayJournal, resolveAllSpeechOverlaps, resolveEntityTimelineLayout, resolveFieldPath, resolveSpeechOverlapByShiftingVideos, safeDurationMs, index_d_exports as schemas, selectAudioScriptSegment, snapshotToPlain, solveVideoDocument, speedOf, syncAggregatedClipsTimePosition, toVideoDocument, validateVideoDocument, videoDocumentMirrorSchema, videoDocumentSchema };
package/dist/index.js CHANGED
@@ -1,4 +1,4 @@
1
- import { $ as findLaneTrack, A as assemblePhoneticScriptContent, B as isMediaAssetVariantKind, C as EntityProjectionGraph, D as ScriptCompositionError, E as decodeEntityRelationRows, F as variantBaseEntityIds, G as toVideoDocument, H as buildSpeechHostMap, I as isJsonObject, J as validateVideoDocument, K as VideoDocumentValidationError, L as createEntityId, M as findComposedAudioScript, N as selectAudioScriptSegment, O as assembleCaptionContent, P as updateEntityFields, Q as ensureLaneTrack, R as createRelationId, S as projectEntityTimeline, St as bytesToBase64, T as EntityTimelineProjectionError, U as derivePositionFromAbs, V as buildInitialVideoDocument, W as fromVideoDocument, X as videoDocumentSchema, Y as partUnionSchema, Z as LANE_KINDS_IN_STACK_ORDER, _ as hasAudioScriptAsset, _t as speedOf, a as createMirrorVideoDocument, at as fillMainTrackTimeGaps, b as assertFieldChanges, bt as videoDocumentMirrorSchema, c as DocumentMutationGuard, ct as resolveAllSpeechOverlaps, d as LoroEntityDraft, dt as TIMELINE_SKELETON_DURATION_MS, et as hasResolvedPlacement, f as readEntityDocument, ft as isEmptyVideoClip, g as audioScriptAssetHash, gt as effectiveVideoClipDurationMs, h as audioScriptAssetFields, ht as DEFAULT_UNIT_TIME_MS, i as MirrorVideoDocumentAdapter, it as arrangeMainTrackSeamlessly, j as assembleScriptText, k as assembleEntityContent, l as LORO_ENTITY_SCHEMA, lt as resolveSpeechOverlapByShiftingVideos, m as audioScriptAssetContent, mt as safeDurationMs, n as createPlainMemoryAdapter, nt as solveVideoDocument, o as createMirrorVideoDocumentAdapter, ot as reassignSpeechesToVideoClipsByTime, p as entityOrderGroups, pt as partDurationMs, q as assertValidVideoDocument, r as generatePartId, rt as cascadeAfterVideoClipChanges, s as readVideoDocumentFromDraft, st as recalculateTimelineDuration, t as PlainMemoryAdapter, tt as laneTrackId, u as LoroEntityDocument, ut as syncAggregatedClipsTimePosition, v as equalJson, vt as partUnionToDraft, w as resolveEntityTimelineLayout, x as resolveFieldPath, xt as base64ToBytes, y as applyFieldChanges, yt as recordEntries, z as hasSequence } from "./document-Bnxm1p5E.js";
1
+ import { $ as findLaneTrack, A as assemblePhoneticScriptContent, B as isMediaAssetVariantKind, C as EntityProjectionGraph, D as ScriptCompositionError, E as decodeEntityRelationRows, F as variantBaseEntityIds, G as toVideoDocument, H as buildSpeechHostMap, I as isJsonObject, J as validateVideoDocument, K as VideoDocumentValidationError, L as createEntityId, M as findComposedAudioScript, N as selectAudioScriptSegment, O as assembleCaptionContent, P as updateEntityFields, Q as ensureLaneTrack, R as createRelationId, S as projectEntityTimeline, St as bytesToBase64, T as EntityTimelineProjectionError, U as derivePositionFromAbs, V as buildInitialVideoDocument, W as fromVideoDocument, X as videoDocumentSchema, Y as partUnionSchema, Z as LANE_KINDS_IN_STACK_ORDER, _ as hasAudioScriptAsset, _t as speedOf, a as createMirrorVideoDocument, at as fillMainTrackTimeGaps, b as assertFieldChanges, bt as videoDocumentMirrorSchema, c as DocumentMutationGuard, ct as resolveAllSpeechOverlaps, d as LoroEntityDraft, dt as TIMELINE_SKELETON_DURATION_MS, et as hasResolvedPlacement, f as readEntityDocument, ft as isEmptyVideoClip, g as audioScriptAssetHash, gt as effectiveVideoClipDurationMs, h as audioScriptAssetFields, ht as DEFAULT_UNIT_TIME_MS, i as MirrorVideoDocumentAdapter, it as arrangeMainTrackSeamlessly, j as assembleScriptText, k as assembleEntityContent, l as LORO_ENTITY_SCHEMA, lt as resolveSpeechOverlapByShiftingVideos, m as audioScriptAssetContent, mt as safeDurationMs, n as createPlainMemoryAdapter, nt as solveVideoDocument, o as createMirrorVideoDocumentAdapter, ot as reassignSpeechesToVideoClipsByTime, p as entityOrderGroups, pt as partDurationMs, q as assertValidVideoDocument, r as generatePartId, rt as cascadeAfterVideoClipChanges, s as readVideoDocumentFromDraft, st as recalculateTimelineDuration, t as PlainMemoryAdapter, tt as laneTrackId, u as LoroEntityDocument, ut as syncAggregatedClipsTimePosition, v as equalJson, vt as partUnionToDraft, w as resolveEntityTimelineLayout, x as resolveFieldPath, xt as base64ToBytes, y as applyFieldChanges, yt as recordEntries, z as hasSequence } from "./document-C30rnewk.js";
2
2
  import { addSpeechesInputSchema, addVideoClipsInputSchema, adjustBgmVolumeInputSchema, adjustSpeechVolumeInputSchema, adjustVideoClipDurationInputSchema, adjustVideoClipVolumeInputSchema, changeSpeechScriptInputSchema, changeSpeechVoiceInputSchema, deleteBgmInputSchema, deleteSpeechesInputSchema, deleteVideoClipsInputSchema, moveSpeechesInputSchema, moveVideoClipsByAnchorInputSchema, moveVideoClipsInputSchema, replaceVideoClipContentInputSchema, replaceVideoClipSequenceInputSchema, setBgmInputSchema, setCaptionStyleInputSchema, setCaptionVisibilityInputSchema, setVideoClipSpeedShiftInputSchema, t as schemas_exports } from "./schemas.js";
3
3
  import { LoroDoc, UndoManager, VersionVector } from "loro-crdt";
4
4
  import { ClientServerSynchronizer, DocManager } from "@mengine/sync";
@@ -579,36 +579,55 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
579
579
  "image",
580
580
  "video",
581
581
  "audio",
582
- "voice"
582
+ "voiceover"
583
583
  ].includes(fact.kind)) throw new Error("Unsupported media kind");
584
584
  if (fact.kind !== "image" && (!Number.isFinite(fact.durationMs) || fact.durationMs <= 0)) throw new Error(`${fact.kind} ${fact.assetId} requires its factual positive duration`);
585
- if ((fact.kind === "audio" || fact.kind === "voice") && fact.storageKey.trim() === "") throw new Error(`${fact.kind} ${fact.assetId} requires its physical storage key`);
585
+ if ((fact.kind === "audio" || fact.kind === "voiceover") && fact.storageKey.trim() === "") throw new Error(`${fact.kind} ${fact.assetId} requires its physical storage key`);
586
586
  decodeEntityRelationRows(initial, numericCoordinates);
587
587
  assertCanonicalEditorResources(initial);
588
588
  const entities = structuredClone([...initial.entities]);
589
589
  const relations = structuredClone([...initial.relations]);
590
- const system = fact.kind === "voice" ? "memota-speech" : "memota";
590
+ const system = fact.kind === "voiceover" ? "memota-speech" : "memota";
591
+ const entityKind = fact.kind === "voiceover" ? "audio" : fact.kind;
592
+ const timbre = fact.kind === "voiceover" ? fact.voice : void 0;
591
593
  const existing = entities.filter((row) => {
592
594
  const external = row.payload.external;
593
595
  return isMediaAssetVariantKind(row.entityKind) && isJsonObject(external) && external.system === system && external.key === fact.assetId;
594
596
  });
597
+ const requireMatchingVariant = (row) => {
598
+ const extent = row.payload.extent;
599
+ if (row.entityKind !== entityKind || fact.kind !== "image" && (!isJsonObject(extent) || extent.start !== 0 || extent.end !== fact.durationMs)) throw new Error(`Media facts conflict with existing identity for ${fact.assetId}`);
600
+ if (fact.storageKey !== void 0 && row.payload.storageKey !== fact.storageKey) {
601
+ if (row.payload.storageKey !== void 0) throw new Error(`Media facts conflict with existing storage key for ${fact.assetId}`);
602
+ return {
603
+ ...row,
604
+ payload: {
605
+ ...row.payload,
606
+ storageKey: fact.storageKey
607
+ }
608
+ };
609
+ }
610
+ return row;
611
+ };
595
612
  if (existing.length > 1) throw new Error(`Duplicate media identities for ${fact.assetId}; reuse ${existing[0].entityId}`);
596
613
  if (existing[0] !== void 0) {
597
- const variant = requireMatchingVariant(existing[0], fact);
614
+ const variant = requireMatchingVariant(existing[0]);
598
615
  replaceEntity$1(entities, variant.entityId, variant);
616
+ const voiceEntityId = bindVoiceTimbre(entities, relations, variant.entityId, timbre, idFactory);
599
617
  return {
600
618
  rows: {
601
619
  entities,
602
620
  relations
603
621
  },
604
- contentEntityId: variant.entityId
622
+ contentEntityId: variant.entityId,
623
+ ...voiceEntityId
605
624
  };
606
625
  }
607
626
  const entityId = createEntityId(idFactory("entity"));
608
627
  if (entities.some((row) => row.entityId === entityId)) throw new Error(`Duplicate entity id ${entityId}`);
609
628
  const variant = {
610
629
  entityId,
611
- entityKind: fact.kind,
630
+ entityKind,
612
631
  payload: {
613
632
  extent: fact.kind === "image" ? {
614
633
  kind: "unbounded",
@@ -624,15 +643,11 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
624
643
  system,
625
644
  key: fact.assetId
626
645
  },
627
- ...fact.storageKey === void 0 ? {} : { storageKey: fact.storageKey },
628
- ...fact.kind === "voice" && fact.voice !== void 0 ? { voice: {
629
- system: fact.voice.system,
630
- key: fact.voice.key,
631
- ...fact.voice.name === void 0 ? {} : { name: fact.voice.name }
632
- } } : {}
646
+ ...fact.storageKey === void 0 ? {} : { storageKey: fact.storageKey }
633
647
  }
634
648
  };
635
649
  entities.push(variant);
650
+ const voiceEntityId = bindVoiceTimbre(entities, relations, entityId, timbre, idFactory);
636
651
  const rows = {
637
652
  entities,
638
653
  relations
@@ -641,33 +656,58 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
641
656
  assertCanonicalEditorResources(rows);
642
657
  return {
643
658
  rows,
644
- contentEntityId: entityId
659
+ contentEntityId: entityId,
660
+ ...voiceEntityId
645
661
  };
646
662
  }
647
- function requireMatchingVariant(row, fact) {
648
- const extent = row.payload.extent;
649
- if (row.entityKind !== fact.kind || fact.kind !== "image" && (!isJsonObject(extent) || extent.start !== 0 || extent.end !== fact.durationMs) || fact.kind === "voice" && !sameVoice(row.payload.voice, fact.voice)) throw new Error(`Media facts conflict with existing identity for ${fact.assetId}`);
650
- if (fact.storageKey !== void 0 && row.payload.storageKey !== fact.storageKey) {
651
- if (row.payload.storageKey !== void 0) throw new Error(`Media facts conflict with existing storage key for ${fact.assetId}`);
652
- return {
653
- ...row,
654
- payload: {
655
- ...row.payload,
656
- storageKey: fact.storageKey
657
- }
658
- };
663
+ /**
664
+ * Attach the timbre a synthesized voiceover was rendered with. One Voice
665
+ * identity is shared by every take naming the same library entry, so an
666
+ * existing row is reused rather than duplicated. Recorded media declares no
667
+ * timbre and keeps no link.
668
+ */
669
+ function bindVoiceTimbre(entities, relations, audioEntityId, descriptor, idFactory) {
670
+ const linked = relations.filter((relation) => relation.relationKind === "voice-timbre" && (relation.endpoint0EntityId === audioEntityId || relation.endpoint1EntityId === audioEntityId));
671
+ if (linked.length > 1) throw new Error(`Voiceover ${audioEntityId} has multiple Voice timbre relations`);
672
+ if (descriptor === void 0) {
673
+ if (linked.length > 0) throw new Error(`Recorded media ${audioEntityId} cannot keep a synthesized Voice timbre`);
674
+ return {};
659
675
  }
660
- return row;
676
+ const voice = entities.find((row) => row.entityKind === "voice" && sameVoiceDescriptor(row.payload.voice, descriptor)) ?? appendVoiceIdentity(entities, descriptor, idFactory);
677
+ const existing = linked[0];
678
+ if (existing === void 0) relations.push({
679
+ relationId: createRelationId(idFactory("relation")),
680
+ relationKind: "voice-timbre",
681
+ endpoint0EntityId: createEntityId(audioEntityId),
682
+ endpoint1EntityId: createEntityId(voice.entityId),
683
+ metadata: {},
684
+ trace: {}
685
+ });
686
+ else if (existing.endpoint0EntityId !== audioEntityId || existing.endpoint1EntityId !== voice.entityId) throw new Error(`Voiceover ${audioEntityId} already carries a different Voice timbre`);
687
+ return { voiceEntityId: voice.entityId };
688
+ }
689
+ function appendVoiceIdentity(entities, descriptor, idFactory) {
690
+ const voice = {
691
+ entityId: createEntityId(idFactory("entity")),
692
+ entityKind: "voice",
693
+ payload: { voice: {
694
+ system: descriptor.system,
695
+ key: descriptor.key,
696
+ ...descriptor.name === void 0 ? {} : { name: descriptor.name }
697
+ } }
698
+ };
699
+ entities.push(voice);
700
+ return voice;
701
+ }
702
+ /** Two takes share a Voice identity only when they name the same library entry. */
703
+ function sameVoiceDescriptor(stored, descriptor) {
704
+ return isJsonObject(stored) && stored.system === descriptor.system && stored.key === descriptor.key && stored.name === descriptor.name;
661
705
  }
662
706
  function replaceEntity$1(entities, entityId, replacement) {
663
707
  const index = entities.findIndex((row) => row.entityId === entityId);
664
708
  if (index < 0) throw new Error(`Entity id ${entityId} does not exist`);
665
709
  entities[index] = replacement;
666
710
  }
667
- function sameVoice(left, right) {
668
- if (right === void 0) return left === void 0;
669
- return isJsonObject(left) && left.system === right.system && left.key === right.key && left.name === right.name && Object.keys(left).every((key) => key === "system" || key === "key" || key === "name");
670
- }
671
711
  //#endregion
672
712
  //#region src/entity-editor/entity-timeline-editor.ts
673
713
  /**
@@ -1203,20 +1243,20 @@ var EntityTimelineEditor = class {
1203
1243
  hostClipEntityId: input.hostClipEntityId,
1204
1244
  anchorOffset: input.anchorOffset
1205
1245
  };
1206
- if (input.placement !== void 0 && (input.hostClipEntityId !== void 0 || input.anchorOffset !== void 0)) throw new Error("Specify Voice placement or hostClipEntityId/anchorOffset, not both");
1246
+ if (input.placement !== void 0 && (input.hostClipEntityId !== void 0 || input.anchorOffset !== void 0)) throw new Error("Specify voiceover placement or hostClipEntityId/anchorOffset, not both");
1207
1247
  requireVolume(input.volume);
1208
1248
  const imported = importMediaAsset(draft, input.media, this.idFactory);
1209
1249
  replaceRows(draft, imported.rows);
1210
- const voice = requireEntityKind(draft, imported.contentEntityId, "voice");
1250
+ const voiceover = requireEntityKind(draft, imported.contentEntityId, "audio");
1211
1251
  const phonetic = this.requireComposedPhoneticScript(draft, input.phoneticScriptEntityId);
1212
- this.ensurePhoneticRender(draft, voice.entityId, phonetic.entityId);
1252
+ this.ensurePhoneticRender(draft, voiceover.entityId, phonetic.entityId);
1213
1253
  const speechTrack = this.ensureRoleTrack(draft, input.timelineEntityId, "speech");
1214
1254
  const existing = draft.entities.find((row) => row.entityId === input.voiceoverClipEntityId);
1215
1255
  let voiceoverClipEntityId;
1216
1256
  if (existing === void 0) voiceoverClipEntityId = this.insertPlacedClipIntoDraft(draft, {
1217
1257
  clipEntityId: input.voiceoverClipEntityId,
1218
1258
  trackEntityId: speechTrack.entityId,
1219
- contentEntityId: voice.entityId,
1259
+ contentEntityId: voiceover.entityId,
1220
1260
  sourceRange: {
1221
1261
  start: 0,
1222
1262
  end: input.media.durationMs
@@ -1240,7 +1280,7 @@ var EntityTimelineEditor = class {
1240
1280
  });
1241
1281
  this.relinkRelation(draft, topology.markerContentRelation, {
1242
1282
  endpoint0EntityId: topology.marker.entityId,
1243
- endpoint1EntityId: voice.entityId
1283
+ endpoint1EntityId: voiceover.entityId
1244
1284
  });
1245
1285
  this.applyClipPlacement(draft, topology.clip.entityId, placement);
1246
1286
  const placed = requireClipTopology(draft, topology.clip.entityId);
@@ -1259,11 +1299,12 @@ var EntityTimelineEditor = class {
1259
1299
  }
1260
1300
  });
1261
1301
  }
1262
- this.reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voice.entityId, phonetic.entityId, input.captions);
1302
+ this.reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voiceover.entityId, phonetic.entityId, input.captions);
1263
1303
  this.resolveOneVoiceoverOverlap(draft);
1264
1304
  return {
1265
1305
  voiceoverClipEntityId,
1266
- voiceEntityId: voice.entityId,
1306
+ voiceoverAudioEntityId: voiceover.entityId,
1307
+ ...imported.voiceEntityId === void 0 ? {} : { voiceEntityId: imported.voiceEntityId },
1267
1308
  phoneticScriptEntityId: phonetic.entityId,
1268
1309
  audioScriptEntityId: requireComposedAudioScript(draft, phonetic.entityId),
1269
1310
  captionClipEntityIds: input.captions.map((caption) => caption.captionClipEntityId)
@@ -1469,23 +1510,23 @@ var EntityTimelineEditor = class {
1469
1510
  draft.relations.push(this.createRelation(draft, "timeline-track", timeline.entityId, track.entityId));
1470
1511
  return track;
1471
1512
  }
1472
- /** The existing pronunciation variant a Voice consumes; composition is required. */
1513
+ /** The existing pronunciation variant a voiceover consumes; composition is required. */
1473
1514
  requireComposedPhoneticScript(draft, phoneticScriptEntityId) {
1474
1515
  const phonetic = requireEntityKind(draft, phoneticScriptEntityId, "phonetic-script");
1475
1516
  requireComposedAudioScript(draft, phonetic.entityId);
1476
1517
  return phonetic;
1477
1518
  }
1478
1519
  /** Preserve the factual synthesis input, regardless of persisted endpoint positions. */
1479
- ensurePhoneticRender(draft, voiceEntityId, phoneticScriptEntityId) {
1480
- const renders = draft.relations.filter((relation) => relation.relationKind === "phonetic-script-render" && (relation.endpoint0EntityId === voiceEntityId || relation.endpoint1EntityId === voiceEntityId));
1481
- if (renders.length > 1) throw new Error(`Voice "${voiceEntityId}" has multiple PhoneticScript render relations`);
1520
+ ensurePhoneticRender(draft, voiceoverAudioEntityId, phoneticScriptEntityId) {
1521
+ const renders = draft.relations.filter((relation) => relation.relationKind === "phonetic-script-render" && (relation.endpoint0EntityId === voiceoverAudioEntityId || relation.endpoint1EntityId === voiceoverAudioEntityId));
1522
+ if (renders.length > 1) throw new Error(`Voiceover "${voiceoverAudioEntityId}" has multiple PhoneticScript render relations`);
1482
1523
  if (renders[0] !== void 0) {
1483
- if (otherEndpoint(renders[0], voiceEntityId) === phoneticScriptEntityId) return;
1484
- throw new Error(`Voice "${voiceEntityId}" already has a different PhoneticScript synthesis input`);
1524
+ if (otherEndpoint(renders[0], voiceoverAudioEntityId) === phoneticScriptEntityId) return;
1525
+ throw new Error(`Voiceover "${voiceoverAudioEntityId}" already has a different PhoneticScript synthesis input`);
1485
1526
  }
1486
- draft.relations.push(this.createRelation(draft, "phonetic-script-render", voiceEntityId, phoneticScriptEntityId));
1527
+ draft.relations.push(this.createRelation(draft, "phonetic-script-render", voiceoverAudioEntityId, phoneticScriptEntityId));
1487
1528
  }
1488
- reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voiceEntityId, phoneticScriptEntityId, captions) {
1529
+ reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voiceoverAudioEntityId, phoneticScriptEntityId, captions) {
1489
1530
  const audioScriptEntityId = requireComposedAudioScript(draft, phoneticScriptEntityId);
1490
1531
  const requested = new Set(captions.map((caption) => caption.captionClipEntityId));
1491
1532
  if (requested.size !== captions.length) throw new Error("Caption Clip ids must be unique within a voiceover take");
@@ -1516,7 +1557,7 @@ var EntityTimelineEditor = class {
1516
1557
  };
1517
1558
  draft.entities.push(content);
1518
1559
  assertCaptionScript(draft, content.entityId, audioScriptEntityId);
1519
- draft.relations.push(this.createRelation(draft, "caption-alignment", content.entityId, voiceEntityId, { alignment: {
1560
+ draft.relations.push(this.createRelation(draft, "caption-alignment", content.entityId, voiceoverAudioEntityId, { alignment: {
1520
1561
  start: startMs,
1521
1562
  end: startMs + durationMs
1522
1563
  } }));
@@ -1573,7 +1614,7 @@ var EntityTimelineEditor = class {
1573
1614
  }
1574
1615
  }
1575
1616
  });
1576
- this.replaceCaptionAlignment(draft, content.entityId, voiceEntityId, startMs, durationMs);
1617
+ this.replaceCaptionAlignment(draft, content.entityId, voiceoverAudioEntityId, startMs, durationMs);
1577
1618
  }
1578
1619
  }
1579
1620
  ensureRoleTrackForClip(draft, clipEntityId, role) {
@@ -1582,10 +1623,10 @@ var EntityTimelineEditor = class {
1582
1623
  const timeline = requireEntityKind(draft, otherEndpoint(requireOneRelation(draft, track.entityId, "timeline-track"), track.entityId), "timeline");
1583
1624
  return this.ensureRoleTrack(draft, timeline.entityId, role);
1584
1625
  }
1585
- replaceCaptionAlignment(draft, captionEntityId, voiceEntityId, startMs, durationMs) {
1626
+ replaceCaptionAlignment(draft, captionEntityId, voiceoverAudioEntityId, startMs, durationMs) {
1586
1627
  const removed = new Set(incidentRelations(draft, captionEntityId).filter((relation) => relation.relationKind === "caption-alignment").map((relation) => relation.relationId));
1587
1628
  draft.relations = draft.relations.filter((relation) => !removed.has(relation.relationId));
1588
- draft.relations.push(this.createRelation(draft, "caption-alignment", captionEntityId, voiceEntityId, { alignment: {
1629
+ draft.relations.push(this.createRelation(draft, "caption-alignment", captionEntityId, voiceoverAudioEntityId, { alignment: {
1589
1630
  start: startMs,
1590
1631
  end: startMs + durationMs
1591
1632
  } }));
@@ -2048,7 +2089,7 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
2048
2089
  relations
2049
2090
  }, {
2050
2091
  assetId: speech.origin_speech_id,
2051
- kind: "voice",
2092
+ kind: "voiceover",
2052
2093
  durationMs: speech.media_duration_ms,
2053
2094
  storageKey: speech.audio_storage_key,
2054
2095
  voice: {
@@ -2136,8 +2177,8 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
2136
2177
  } else if (part.caption !== void 0) {
2137
2178
  const caption = part.caption;
2138
2179
  duration = caption.initial_duration_ms;
2139
- const voice = contentByPartId.get(caption.speech_part_id);
2140
- const render = relations.find((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voice);
2180
+ const voiceover = contentByPartId.get(caption.speech_part_id);
2181
+ const render = relations.find((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voiceover);
2141
2182
  const phonetic = entities.find((entity) => entity.entityId === render?.endpoint1EntityId);
2142
2183
  const script = phonetic === void 0 ? void 0 : composedAudioScript(phonetic, entities);
2143
2184
  if (script === void 0) fail("Legacy Caption has no composed AudioScript");
@@ -2164,7 +2205,7 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
2164
2205
  selection,
2165
2206
  ...caption.style === void 0 ? {} : { style: toEntityCaptionStyle(caption.style) }
2166
2207
  });
2167
- link("caption-alignment", content, voice, { alignment: {
2208
+ link("caption-alignment", content, voiceover, { alignment: {
2168
2209
  start: caption.start_ms,
2169
2210
  end: caption.start_ms + duration
2170
2211
  } });
@@ -2503,21 +2544,21 @@ function requirePart(document, partId) {
2503
2544
  if (part === void 0) fail(`Placed Clip ${partId} has no part payload`);
2504
2545
  return part;
2505
2546
  }
2506
- function ensureSpeechPhoneticChain(voiceEntityId, audioScriptText, phonemeScript, entities, relations, addEntity, link, documentAudioScriptEntityId) {
2507
- const renders = relations.filter((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voiceEntityId);
2508
- if (renders.length > 1) fail(`Voice ${voiceEntityId} has ambiguous PhoneticScript provenance`);
2547
+ function ensureSpeechPhoneticChain(voiceoverAudioEntityId, audioScriptText, phonemeScript, entities, relations, addEntity, link, documentAudioScriptEntityId) {
2548
+ const renders = relations.filter((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voiceoverAudioEntityId);
2549
+ if (renders.length > 1) fail(`Voiceover ${voiceoverAudioEntityId} has ambiguous PhoneticScript provenance`);
2509
2550
  if (renders.length === 1) {
2510
2551
  const phonetic = entities.find((entity) => entity.entityId === renders[0].endpoint1EntityId && entity.entityKind === "phonetic-script");
2511
- if (phonetic === void 0) fail(`Voice ${voiceEntityId} has a dangling PhoneticScript render`);
2552
+ if (phonetic === void 0) fail(`Voiceover ${voiceoverAudioEntityId} has a dangling PhoneticScript render`);
2512
2553
  const script = composedAudioScript(phonetic, entities);
2513
- if (script === void 0 || scriptText(script) !== audioScriptText) fail(`Voice ${voiceEntityId} conflicts with the legacy Speech script`);
2554
+ if (script === void 0 || scriptText(script) !== audioScriptText) fail(`Voiceover ${voiceoverAudioEntityId} conflicts with the legacy Speech script`);
2514
2555
  return phonetic.entityId;
2515
2556
  }
2516
2557
  const script = entities.find((entity) => entity.entityId === documentAudioScriptEntityId);
2517
2558
  if (script === void 0 || script.entityKind !== "audio-script") fail("Migration must carry the fixed document AudioScript");
2518
2559
  const segments = script.payload.segments;
2519
2560
  if (Array.isArray(segments) && segments.length > 0) {
2520
- if (scriptText(script) !== audioScriptText) fail(`Voice ${voiceEntityId} conflicts with the project AudioScript text`);
2561
+ if (scriptText(script) !== audioScriptText) fail(`Voiceover ${voiceoverAudioEntityId} conflicts with the project AudioScript text`);
2521
2562
  } else {
2522
2563
  const scriptIndex = entities.findIndex((row) => row.entityId === script.entityId);
2523
2564
  entities[scriptIndex] = {
@@ -2532,7 +2573,7 @@ function ensureSpeechPhoneticChain(voiceEntityId, audioScriptText, phonemeScript
2532
2573
  baseEntityIds: [script.entityId],
2533
2574
  ...phonemeScript === void 0 ? {} : { phonemeScript }
2534
2575
  });
2535
- link("phonetic-script-render", voiceEntityId, phonetic);
2576
+ link("phonetic-script-render", voiceoverAudioEntityId, phonetic);
2536
2577
  return phonetic;
2537
2578
  }
2538
2579
  function composedAudioScript(phonetic, entities) {
package/dist/testing.js CHANGED
@@ -1,4 +1,4 @@
1
- import { St as bytesToBase64, o as createMirrorVideoDocumentAdapter, xt as base64ToBytes } from "./document-Bnxm1p5E.js";
1
+ import { St as bytesToBase64, o as createMirrorVideoDocumentAdapter, xt as base64ToBytes } from "./document-C30rnewk.js";
2
2
  import { t as classifyUpdate } from "./loro-relay-doc-cJSY-uau.js";
3
3
  import { LoroDoc, VersionVector, encodeFrontiers } from "loro-crdt";
4
4
  //#region src/testing/in-memory-mengine-server.ts
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@mengine/medeo-client",
3
- "version": "2.0.1-alpha.7",
3
+ "version": "2.0.1-alpha.9",
4
4
  "license": "UNLICENSED",
5
5
  "repository": {
6
6
  "type": "git",
@@ -28,9 +28,9 @@
28
28
  "immer": "^10.2.0",
29
29
  "loro-mirror": "^2.3.3",
30
30
  "zod": "^4.4.3",
31
- "@mengine/storage": "2.0.1-alpha.7",
32
- "@mengine/sync": "2.0.1-alpha.7",
33
- "@mengine/utils": "2.0.1-alpha.7"
31
+ "@mengine/storage": "2.0.1-alpha.9",
32
+ "@mengine/sync": "2.0.1-alpha.9",
33
+ "@mengine/utils": "2.0.1-alpha.9"
34
34
  },
35
35
  "devDependencies": {
36
36
  "@types/node": "^25.9.1",
@@ -40,8 +40,8 @@
40
40
  "typescript": "^6.0.3",
41
41
  "vite-plugin-wasm": "^3.6.0",
42
42
  "vite-plus": "^0.1.23",
43
- "@mengine/idl-codegen": "0.0.0",
44
- "@mengine/medeo-dsl": "0.0.0"
43
+ "@mengine/medeo-dsl": "0.0.0",
44
+ "@mengine/idl-codegen": "0.0.0"
45
45
  },
46
46
  "peerDependencies": {
47
47
  "loro-crdt": "^1.16.1"