@mengine/medeo-client 2.0.1-alpha.7 → 2.0.1-alpha.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1447,8 +1447,63 @@ function emptyLaneTracks() {
1447
1447
  }
1448
1448
  //#endregion
1449
1449
  //#region ../medeo-dsl/src/entities.ts
1450
+ /**
1451
+ * The declared trait set of every known kind.
1452
+ *
1453
+ * `caption` carries `sequence` on purpose: a Caption owns the display timing of
1454
+ * the one AudioScript segment it shows, which is what admits it into a Clip.
1455
+ * `voice` carries no media or sequence trait — it is a timbre identity, not
1456
+ * playable content; a rendered voiceover is an `audio` entity.
1457
+ */
1458
+ const ENTITY_TRAIT_KINDS = Object.freeze({
1459
+ axvideo: Object.freeze([
1460
+ "container",
1461
+ "sequence",
1462
+ "visual",
1463
+ "audible"
1464
+ ]),
1465
+ timeline: Object.freeze(["container"]),
1466
+ track: Object.freeze(["container", "sequence"]),
1467
+ clip: Object.freeze(["container"]),
1468
+ asset: Object.freeze([]),
1469
+ video: Object.freeze([
1470
+ "media-file",
1471
+ "sequence",
1472
+ "visual",
1473
+ "audible"
1474
+ ]),
1475
+ audio: Object.freeze([
1476
+ "media-file",
1477
+ "sequence",
1478
+ "audible"
1479
+ ]),
1480
+ image: Object.freeze([
1481
+ "media-file",
1482
+ "sequence",
1483
+ "visual"
1484
+ ]),
1485
+ voice: Object.freeze(["container", "ip-bound"]),
1486
+ "sequence-marker": Object.freeze([]),
1487
+ viewport: Object.freeze([]),
1488
+ "audio-script": Object.freeze(["segment-text"]),
1489
+ "phonetic-script": Object.freeze(["segment-text"]),
1490
+ caption: Object.freeze(["segment-text", "sequence"])
1491
+ });
1492
+ /** Declared traits of a known kind; extension kinds declare none through this table. */
1493
+ function traitKindsOf(entityKind) {
1494
+ return isKnownEntityKind(entityKind) ? ENTITY_TRAIT_KINDS[entityKind] : [];
1495
+ }
1496
+ function hasTraitKind(entityKind, trait) {
1497
+ return traitKindsOf(entityKind).includes(trait);
1498
+ }
1499
+ /**
1500
+ * Sequence admission asks the declared traits first, so a kind that does not
1501
+ * declare `sequence` can never buy its way in by carrying the fields. Extension
1502
+ * kinds declare no traits and stay structurally detected.
1503
+ */
1450
1504
  function hasSequence(entity) {
1451
- if (isKnownNonSequenceKind(entity.entityKind) || isReservedEntityKind(entity.entityKind)) return false;
1505
+ if (isReservedEntityKind(entity.entityKind)) return false;
1506
+ if (isKnownEntityKind(entity.entityKind) && !hasTraitKind(entity.entityKind, "sequence")) return false;
1452
1507
  return isSequenceFields(entity);
1453
1508
  }
1454
1509
  function isSequenceFields(value) {
@@ -1461,22 +1516,16 @@ function isSequenceFields(value) {
1461
1516
  if (value.sampling !== "native" && value.sampling !== "constant" && value.sampling !== "derived") return false;
1462
1517
  return "coordinateSpace" in value && value.coordinateSpace !== void 0;
1463
1518
  }
1464
- function isKnownSequenceKind(kind) {
1465
- return isMediaAssetVariantKind(kind) || kind === "caption" || kind === "axvideo";
1466
- }
1467
1519
  function isMediaAssetVariantKind(kind) {
1468
- return kind === "video" || kind === "image" || kind === "audio" || kind === "voice";
1520
+ return kind === "video" || kind === "image" || kind === "audio";
1469
1521
  }
1470
1522
  function isKnownEntityKind(kind) {
1471
- return isKnownSequenceKind(kind) || isKnownNonSequenceKind(kind);
1523
+ return Object.hasOwn(ENTITY_TRAIT_KINDS, kind);
1472
1524
  }
1473
1525
  function isReservedEntityKind(kind) {
1474
1526
  const normalized = kind.toLowerCase().replaceAll("-", "").replaceAll("_", "");
1475
1527
  return normalized === "speech" || normalized === "videodocument";
1476
1528
  }
1477
- function isKnownNonSequenceKind(kind) {
1478
- return kind === "timeline" || kind === "track" || kind === "clip" || kind === "asset" || kind === "sequence-marker" || kind === "viewport" || kind === "audio-script" || kind === "phonetic-script";
1479
- }
1480
1529
  function isRecord$1(value) {
1481
1530
  return typeof value === "object" && value != null && !Array.isArray(value);
1482
1531
  }
@@ -2059,12 +2108,15 @@ function validateKnownEntityPayload(entity) {
2059
2108
  break;
2060
2109
  case "video":
2061
2110
  case "audio":
2062
- case "voice":
2111
+ validateSequencePayload(value, "bounded", "native", problems);
2112
+ validateMediaAssetFields(value, problems);
2113
+ break;
2063
2114
  case "caption":
2064
2115
  validateSequencePayload(value, "bounded", "native", problems);
2065
- if (entity.entityKind !== "caption") validateMediaAssetFields(value, problems);
2066
- if (entity.entityKind === "voice") validateVoicePayload(value, problems);
2067
- if (entity.entityKind === "caption") validateCaptionPayload(value, problems);
2116
+ validateCaptionPayload(value, problems);
2117
+ break;
2118
+ case "voice":
2119
+ validateVoicePayload(value, problems);
2068
2120
  break;
2069
2121
  case "image":
2070
2122
  validateSequencePayload(value, "unbounded", "constant", problems);
@@ -2171,11 +2223,26 @@ function validateExternalLocator(value, problems) {
2171
2223
  function validateStorageKey(value, problems) {
2172
2224
  if (value.storageKey !== void 0 && (typeof value.storageKey !== "string" || value.storageKey.trim() === "")) problems.push("storageKey must be a non-empty string when present");
2173
2225
  }
2226
+ /**
2227
+ * Voice is a timbre identity, not playable content: it owns the voice-library
2228
+ * descriptor and nothing that would make it look like media or a Sequence. The
2229
+ * rendered voiceover is an `audio` entity bound by a `voice-timbre` Relation.
2230
+ */
2174
2231
  function validateVoicePayload(value, problems) {
2232
+ for (const key of [
2233
+ "extent",
2234
+ "sampling",
2235
+ "coordinateSpace",
2236
+ "external",
2237
+ "storageKey"
2238
+ ]) if (Object.hasOwn(value, key)) problems.push(`${key} belongs to the rendered Audio, not to the Voice identity`);
2175
2239
  const voice = value.voice;
2176
- if (voice === void 0) return;
2240
+ if (voice === void 0) {
2241
+ problems.push("voice is required; a Voice entity is its voice-library identity");
2242
+ return;
2243
+ }
2177
2244
  if (!isRecord(voice)) {
2178
- problems.push("voice must be an object when present");
2245
+ problems.push("voice must be an object");
2179
2246
  return;
2180
2247
  }
2181
2248
  if (voice.system !== "voice-library") problems.push("voice.system must be \"voice-library\"");
@@ -2307,7 +2374,6 @@ function isOwnedLocalEntityValuePath(entityKind, path) {
2307
2374
  if ([
2308
2375
  "video",
2309
2376
  "audio",
2310
- "voice",
2311
2377
  "image",
2312
2378
  "caption",
2313
2379
  "axvideo"
@@ -2371,7 +2437,7 @@ function collectRelatedEntities(entityRefs, index) {
2371
2437
  };
2372
2438
  }
2373
2439
  function expectedSequenceShape(kind) {
2374
- if (kind === "video" || kind === "audio" || kind === "voice" || kind === "caption") return {
2440
+ if (kind === "video" || kind === "audio" || kind === "caption") return {
2375
2441
  extent: "bounded",
2376
2442
  sampling: "native"
2377
2443
  };
@@ -2412,7 +2478,7 @@ const generatedRelationSpec = Object.freeze({
2412
2478
  });
2413
2479
  const captionAlignmentRelationSpec = Object.freeze({
2414
2480
  kind: "caption-alignment",
2415
- validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["caption"]), new Set(["audio", "voice"])),
2481
+ validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["caption"]), new Set(["audio"])),
2416
2482
  validateMetadata: isCaptionAlignmentMetadata
2417
2483
  });
2418
2484
  /** `clip-anchor(child, host)` means endpoint 0 follows endpoint 1. */
@@ -2421,20 +2487,29 @@ const clipAnchorRelationSpec = Object.freeze({
2421
2487
  validateEndpoints: (endpoints) => endpoints[0].current().entityKind === "clip" && endpoints[1].current().entityKind === "clip",
2422
2488
  validateMetadata: isEmptyMetadata
2423
2489
  });
2424
- /** Voice is generated from PhoneticScript; kinds determine roles regardless of endpoint positions. */
2490
+ /**
2491
+ * The rendered voiceover Audio and the PhoneticScript it was synthesized from;
2492
+ * kinds determine roles regardless of endpoint positions.
2493
+ */
2425
2494
  const phoneticScriptRenderRelationSpec = Object.freeze({
2426
2495
  kind: "phonetic-script-render",
2427
- validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["voice"]), new Set(["phonetic-script"])),
2496
+ validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio"]), new Set(["phonetic-script"])),
2497
+ validateMetadata: isEmptyMetadata
2498
+ });
2499
+ /**
2500
+ * The timbre identity a rendered voiceover Audio was synthesized with. Voice
2501
+ * stays an identity: it never carries the audio itself, so the link between the
2502
+ * two is a Relation rather than one shared entity.
2503
+ */
2504
+ const voiceTimbreRelationSpec = Object.freeze({
2505
+ kind: "voice-timbre",
2506
+ validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio"]), new Set(["voice"])),
2428
2507
  validateMetadata: isEmptyMetadata
2429
2508
  });
2430
2509
  /** AudioScript was transcribed from Audio, Video, or recorded Voice; kinds determine roles. */
2431
2510
  const audioScriptSourceRelationSpec = Object.freeze({
2432
2511
  kind: "audio-script-source",
2433
- validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio-script"]), new Set([
2434
- "audio",
2435
- "video",
2436
- "voice"
2437
- ])),
2512
+ validateEndpoints: (endpoints) => hasKinds(endpoints, new Set(["audio-script"]), new Set(["audio", "video"])),
2438
2513
  validateMetadata: isEmptyMetadata
2439
2514
  });
2440
2515
  /**
@@ -2461,6 +2536,7 @@ const builtInRelationSpecs = Object.freeze([
2461
2536
  captionAlignmentRelationSpec,
2462
2537
  clipAnchorRelationSpec,
2463
2538
  phoneticScriptRenderRelationSpec,
2539
+ voiceTimbreRelationSpec,
2464
2540
  audioScriptSourceRelationSpec,
2465
2541
  audioScriptMarkerRelationSpec
2466
2542
  ]);
@@ -2490,7 +2566,7 @@ function hasAssetAndSequence(endpoints) {
2490
2566
  return first.entityKind === "asset" && hasSequence(second) || second.entityKind === "asset" && hasSequence(first);
2491
2567
  }
2492
2568
  function isGeneratedMedia(entity) {
2493
- return entity.entityKind === "video" || entity.entityKind === "image" || entity.entityKind === "audio" || entity.entityKind === "voice";
2569
+ return entity.entityKind === "video" || entity.entityKind === "image" || entity.entityKind === "audio";
2494
2570
  }
2495
2571
  function isEmptyMetadata(value) {
2496
2572
  return isJsonObject(value) && Object.keys(value).length === 0;
@@ -2920,7 +2996,7 @@ function resolveGraphLayout(graph) {
2920
2996
  if (clip.entityKind !== "clip" || graph.adjacent(clip.entityId, "track-clip").length !== 1) fail(`Clip ${clip.entityId} must belong to exactly one Track`);
2921
2997
  const marker = graph.one(clip.entityId, "clip-marker", "sequence-marker");
2922
2998
  const content = graph.one(marker.entityId, "marker-content");
2923
- if (!(typedRole === "video_clip" ? ["image", "video"] : typedRole === "speech" ? ["voice"] : typedRole === "caption" ? ["caption"] : ["audio"]).includes(content.entityKind)) fail(`unsupported compatibility content kind ${content.entityKind} on ${typedRole} Track`);
2999
+ if (!(typedRole === "video_clip" ? ["image", "video"] : typedRole === "speech" ? ["audio"] : typedRole === "caption" ? ["caption"] : ["audio"]).includes(content.entityKind)) fail(`unsupported compatibility content kind ${content.entityKind} on ${typedRole} Track`);
2924
3000
  const space = content.payload.coordinateSpace;
2925
3001
  if (space !== "ms" && space !== "milliseconds" && (!object(space) || space.unit !== "ms" && space.unit !== "milliseconds")) fail(`media ${content.entityId} must state millisecond coordinates for this reader`);
2926
3002
  const source = range(marker.payload.sourceRange, "sourceRange");
@@ -2999,7 +3075,7 @@ function resolveGraphLayout(graph) {
2999
3075
  let start = fact.absoluteStart ?? 0;
3000
3076
  if (fact.mode === "anchored") {
3001
3077
  const host = resolve(fact.anchorId);
3002
- if (fact.role === "speech" && !["video_clip", "bgm"].includes(facts.get(host.clipEntityId).role)) fail("An anchored Voice must use a visual or BGM Clip as its host");
3078
+ if (fact.role === "speech" && !["video_clip", "bgm"].includes(facts.get(host.clipEntityId).role)) fail("An anchored voiceover must use a visual or BGM Clip as its host");
3003
3079
  start = host.start + fact.anchorOffset;
3004
3080
  } else if (fact.predecessor !== void 0) {
3005
3081
  const prior = resolve(fact.predecessor);
@@ -3083,8 +3159,11 @@ function projectEntityTimeline(rows) {
3083
3159
  const asset = mediaAssetFacts(graph, content);
3084
3160
  requireUntrimmedAudio(content, marker, placement);
3085
3161
  const external = externalLocator(asset, "memota-speech");
3086
- const assembled = graph.adjacent(content.entityId, "phonetic-script-render").length > 0 || content.payload.voice !== void 0 ? assemblePhoneticScriptContent(rows, graph.one(content.entityId, "phonetic-script-render", "phonetic-script").entityId) : void 0;
3087
- const voice = content.payload.voice;
3162
+ const renderEdges = graph.adjacent(content.entityId, "phonetic-script-render");
3163
+ const timbreEdges = graph.adjacent(content.entityId, "voice-timbre");
3164
+ const assembled = renderEdges.length > 0 || timbreEdges.length > 0 ? assemblePhoneticScriptContent(rows, graph.one(content.entityId, "phonetic-script-render", "phonetic-script").entityId) : void 0;
3165
+ if (timbreEdges.length > 1) fail("A voiceover resolves at most one Voice identity");
3166
+ const voice = timbreEdges.length === 0 ? void 0 : graph.one(content.entityId, "voice-timbre", "voice").payload.voice;
3088
3167
  if (voice !== void 0 && (!object(voice) || voice.system !== "voice-library" || typeof voice.key !== "string")) fail("Voice must use an explicit voice-library locator");
3089
3168
  const captionIds = layout.tracks.flatMap((candidate) => candidate.role === "caption" ? candidate.clips.filter((caption) => caption.anchorClipEntityId === clip.entityId).map((caption) => caption.clipEntityId) : []);
3090
3169
  parts[clip.entityId] = { speech: {
package/dist/index.d.ts CHANGED
@@ -272,17 +272,22 @@ interface AudioMediaAssetFact {
272
272
  readonly durationMs: number;
273
273
  readonly storageKey: string;
274
274
  }
275
- interface VoiceMediaAssetFact {
275
+ /**
276
+ * A speech recording. It materializes as an `audio` entity like every other
277
+ * playable sound; `voice` names the timbre identity it was synthesized with,
278
+ * which becomes its own Voice entity linked by a `voice-timbre` Relation.
279
+ */
280
+ interface VoiceoverMediaAssetFact {
276
281
  /** Stable external speech result id, independent of the placed Clip id. */
277
282
  readonly assetId: string;
278
- readonly kind: 'voice';
283
+ readonly kind: 'voiceover';
279
284
  readonly durationMs: number;
280
285
  readonly storageKey: string;
281
- /** Present for synthesized voice, absent for original recorded audio. */
286
+ /** Present for synthesized speech, absent for original recorded audio. */
282
287
  readonly voice?: VoiceDescriptor;
283
288
  }
284
289
  /** Facts resolved from media storage. A trim window never substitutes for intrinsic duration. */
285
- type MediaAssetFact = ImageMediaAssetFact | VideoMediaAssetFact | AudioMediaAssetFact | VoiceMediaAssetFact;
290
+ type MediaAssetFact = ImageMediaAssetFact | VideoMediaAssetFact | AudioMediaAssetFact | VoiceoverMediaAssetFact;
286
291
  type ClipPlacement = {
287
292
  readonly kind: 'sequential';
288
293
  readonly order: number;
@@ -334,7 +339,7 @@ interface DeleteClipTreeInput {
334
339
  interface VoiceoverCaptionFact {
335
340
  /** Stable placed caption identity supplied by the materialized side effect. */
336
341
  readonly captionClipEntityId: string;
337
- /** Directly held bases; includes the AudioScript used by the Voice. */
342
+ /** Directly held bases; includes the AudioScript the voiceover was rendered from. */
338
343
  readonly baseEntityIds: readonly string[];
339
344
  /** Ordered selection of AudioScript segments; caption text is never passed inline. */
340
345
  readonly selection: CaptionSegmentSelection;
@@ -345,7 +350,7 @@ interface VoiceoverCaptionFact {
345
350
  type VoiceoverTakeInput = {
346
351
  readonly timelineEntityId: string; /** Stable placed speech identity, distinct from media.assetId. */
347
352
  readonly voiceoverClipEntityId: string;
348
- readonly media: VoiceMediaAssetFact; /** Existing pronunciation variant; its composed AudioScript stays the text owner. */
353
+ readonly media: VoiceoverMediaAssetFact; /** Existing pronunciation variant; its composed AudioScript stays the text owner. */
349
354
  readonly phoneticScriptEntityId: string;
350
355
  readonly volume: number;
351
356
  readonly captions: readonly VoiceoverCaptionFact[];
@@ -421,8 +426,11 @@ interface TrimClipInput {
421
426
  }
422
427
  interface VoiceoverTakeResult {
423
428
  readonly voiceoverClipEntityId: string;
424
- readonly voiceEntityId: string;
425
- /** The pronunciation variant the Voice was rendered from. */
429
+ /** The rendered voiceover Audio entity. */
430
+ readonly voiceoverAudioEntityId: string;
431
+ /** The timbre identity it was synthesized with, when the take declared one. */
432
+ readonly voiceEntityId?: string;
433
+ /** The pronunciation variant the voiceover was rendered from. */
426
434
  readonly phoneticScriptEntityId: string;
427
435
  /** The base-text owner resolved from the PhoneticScript baseEntityIds. */
428
436
  readonly audioScriptEntityId: string;
@@ -556,7 +564,7 @@ declare class EntityTimelineEditor {
556
564
  insertCaptionClip(input: InsertCaptionClipInput): ClipEntityId;
557
565
  private insertPlacedClipIntoDraft;
558
566
  private ensureRoleTrack;
559
- /** The existing pronunciation variant a Voice consumes; composition is required. */
567
+ /** The existing pronunciation variant a voiceover consumes; composition is required. */
560
568
  private requireComposedPhoneticScript;
561
569
  /** Preserve the factual synthesis input, regardless of persisted endpoint positions. */
562
570
  private ensurePhoneticRender;
@@ -632,6 +640,8 @@ interface ImportedMediaAsset {
632
640
  readonly rows: EntityRelationRows;
633
641
  /** The single media variant identity for this asset id; it is its own Asset. */
634
642
  readonly contentEntityId: string;
643
+ /** The timbre identity of a synthesized voiceover; absent for recorded media. */
644
+ readonly voiceEntityId?: string;
635
645
  }
636
646
  /**
637
647
  * Import at the editor boundary, not at asset generation time. Existing
@@ -1716,4 +1726,4 @@ declare function hasAudioScriptAsset(payload: JsonObject): boolean;
1716
1726
  /** Internal fingerprint prevents a resource from silently describing an older text revision. */
1717
1727
  declare function audioScriptAssetFields(payload: JsonObject, assetId: string): JsonObject;
1718
1728
  //#endregion
1719
- export { type Aggregation, type AssembledCaptionContent, type AssembledPhoneticScriptContent, type Attachment, type AudioMediaAssetFact, type BgmPart, type CaptionContentInput, type CaptionDisplayCue, type CaptionPart, type CaptionSegmentSelection, type CaptionStyle, type ClipEntityId, type ClipPlacement, type CommitOptions, DEFAULT_UNIT_TIME_MS, type DeleteBgmInput, type DeleteClipInput, type DeleteClipTreeInput, type DeleteVoiceoverInput, type DerivedItemPosition, type DocStorageLike, type DocVersionMark, type DocumentAudioScript, type DocumentAudioScriptState, type EditorFoundation, EntityDocumentState, type EntityGraphCommitOptions, EntityGraphHttpClient, type EntityGraphState, EntityTimelineEditor, type EntityTimelineIdFactory, EntityTimelineProjectionError, type EntityTimelineTrackRole, type FieldChange, type FieldPath, IMPLEMENTED_SEMANTIC_OP_KINDS, type ImageMediaAssetFact, type ImplementedSemanticOpKind, type ImportedMediaAsset, type InsertCaptionClipInput, type InsertClipInput, type InsertMediaClipInput, type InsertMediaClipsInput, type InsertPlacedClipInput, type JournalEntry, LANE_KINDS_IN_STACK_ORDER, LORO_ENTITY_SCHEMA, type LaneKind, type LinearClipSpeed, LoroEntityDocument, LoroEntityDraft, type MainClipRange, type MakeEmptyPart, ManualSyncDoc, type ManualSyncDocOptions, MedeoHttpDocStorage, type MedeoHttpDocStorageOptions, type MediaAssetFact, type MediaClipInsertion, MemoryDocStorage, MengineAckFailedError, type MengineAuditEntry, type MengineAuditResponse, MengineDocSession, type MengineDocSessionOptions, type MengineDocSessionUpdateEvent, type MengineDocSyncState, type MengineDocumentVersion, type MengineEventStreamOptions, MengineHttpClient, type MengineHttpClientOptions, MengineHttpRequestError, MenginePushRejectedError, type MenginePushResponse, type MenginePushUpdateResponse, type MengineRejectedResponse, type MengineSnapshotResponse, type MengineSseUpdateEvent, type MengineSyncResponse, type MengineUndoState, type MengineUpdateMeta, MirrorVideoDocumentAdapter, type MirrorVideoDocumentOptions, type MoveClipInput, type MoveClipsToStartsInput, type MoveSequentialClipsInput, type MoveVoiceoverInput, type OpActor, type PartAggregation, type PartIdFactory, type PartKind, type PartUnion, type PatchCaptionStyleInput, PlainMemoryAdapter, type PlainMemoryAdapterOptions, type PlannedSemanticOpKind, type PullFailureReason, type PullResult, type PushOutcome, type PushOutcomeKind, type PushResult, type PushResultKind, type ReplaceClipContentInput, type ReplaceMediaClipInput, type ReplaceSequentialClipsInput, type ReplacementMediaClipInput, type ResolvedEntityClipPlacement, type ResolvedEntityTimelineLayout, type ResolvedEntityTrack, SchemaValidator, ScriptCompositionError, type SemanticDocumentAdapter, SemanticEditor, type SemanticOpInput, type SemanticOpKind, type SemanticOpName, type SequentialClipAnchor, type SetBgmInput, type SetCaptionVisibilityInput, type SetClipPlacementInput, type SetClipSpeedInput, type SetClipVolumeInput, type SnapshotReadable, type SolvedVideoDocument, type SpeechHostMap, type SpeechPart, type SpeedShift, TIMELINE_SKELETON_DURATION_MS, type Timeline, type TimelineDoc, type TimelineItem, type Track, type TrackDraft, type TrackItem, type TrackItemDraft, type TrackItemTimePosition, type TransactAudit, type TrimClipInput, type UpdateClipInput, type UpdateClipMarkerInput, ValidationError, type VideoClipPart, type VideoDocument, type VideoDocumentDraft, type VideoDocumentMirrorSchema, VideoDocumentValidationError, type VideoDocumentValidationIssue, type VideoDocumentValidationIssueCode, type VideoDraft, type CaptionPart$1 as VideoDraftCaptionPart, type VideoDraftContent, type VideoDraftPartUnion, type Timeline$1 as VideoDraftTimeline, type Track$1 as VideoDraftTrack, type TrackItem$1 as VideoDraftTrackItem, type VideoMediaAssetFact, type VisualMediaAssetFact, type VoiceMediaAssetFact, type VoiceoverCaptionFact, type VoiceoverTakeInput, type VoiceoverTakeResult, type WaitForServerAckOptions, applyEntityRows, applyFieldChanges, arrangeMainTrackSeamlessly, assembleCaptionContent, assembleEntityContent, assemblePhoneticScriptContent, assembleScriptText, assertCanonicalEditorResources, assertEntityProjectionPreservesLegacy, assertFieldChanges, assertMediaAssetWritePolicy, assertValidVideoDocument, audioScriptAssetContent, audioScriptAssetFields, audioScriptAssetHash, base64ToBytes, buildInitialVideoDocument, buildSpeechHostMap, bytesToBase64, cascadeAfterVideoClipChanges, compileEntityRows, createEditSandbox, createMirrorVideoDocument, createMirrorVideoDocumentAdapter, createPlainMemoryAdapter, decodeDocVersionMark, derivePositionFromAbs, effectiveVideoClipDurationMs, encodeDocVersionMark, ensureEditorFoundation, ensureLaneTrack, fillMainTrackTimeGaps, findComposedAudioScript, findLaneTrack, fromVideoDocument, generatePartId, getAt, hasAudioScriptAsset, hostForAbsMs, importMediaAsset, isEmptyVideoClip, isImplementedSemanticOpKind, isMap, laneTrackId, mainTrackRanges, migrateLegacyTimelineToEntities, partDurationMs, partUnionSchema, prepareCaptionContent, projectEntityTimeline, readDocumentAudioScript, readEntityDocument, readMainTrackItems, readMengineEventStream, readPart, readPartDurationMs, readVideoDocumentFromDraft, reassignSpeechesToVideoClipsByTime, recalculateTimelineDuration, relativePositionForAbs, replayJournal, resolveAllSpeechOverlaps, resolveEntityTimelineLayout, resolveFieldPath, resolveSpeechOverlapByShiftingVideos, safeDurationMs, index_d_exports as schemas, selectAudioScriptSegment, snapshotToPlain, solveVideoDocument, speedOf, syncAggregatedClipsTimePosition, toVideoDocument, validateVideoDocument, videoDocumentMirrorSchema, videoDocumentSchema };
1729
+ export { type Aggregation, type AssembledCaptionContent, type AssembledPhoneticScriptContent, type Attachment, type AudioMediaAssetFact, type BgmPart, type CaptionContentInput, type CaptionDisplayCue, type CaptionPart, type CaptionSegmentSelection, type CaptionStyle, type ClipEntityId, type ClipPlacement, type CommitOptions, DEFAULT_UNIT_TIME_MS, type DeleteBgmInput, type DeleteClipInput, type DeleteClipTreeInput, type DeleteVoiceoverInput, type DerivedItemPosition, type DocStorageLike, type DocVersionMark, type DocumentAudioScript, type DocumentAudioScriptState, type EditorFoundation, EntityDocumentState, type EntityGraphCommitOptions, EntityGraphHttpClient, type EntityGraphState, EntityTimelineEditor, type EntityTimelineIdFactory, EntityTimelineProjectionError, type EntityTimelineTrackRole, type FieldChange, type FieldPath, IMPLEMENTED_SEMANTIC_OP_KINDS, type ImageMediaAssetFact, type ImplementedSemanticOpKind, type ImportedMediaAsset, type InsertCaptionClipInput, type InsertClipInput, type InsertMediaClipInput, type InsertMediaClipsInput, type InsertPlacedClipInput, type JournalEntry, LANE_KINDS_IN_STACK_ORDER, LORO_ENTITY_SCHEMA, type LaneKind, type LinearClipSpeed, LoroEntityDocument, LoroEntityDraft, type MainClipRange, type MakeEmptyPart, ManualSyncDoc, type ManualSyncDocOptions, MedeoHttpDocStorage, type MedeoHttpDocStorageOptions, type MediaAssetFact, type MediaClipInsertion, MemoryDocStorage, MengineAckFailedError, type MengineAuditEntry, type MengineAuditResponse, MengineDocSession, type MengineDocSessionOptions, type MengineDocSessionUpdateEvent, type MengineDocSyncState, type MengineDocumentVersion, type MengineEventStreamOptions, MengineHttpClient, type MengineHttpClientOptions, MengineHttpRequestError, MenginePushRejectedError, type MenginePushResponse, type MenginePushUpdateResponse, type MengineRejectedResponse, type MengineSnapshotResponse, type MengineSseUpdateEvent, type MengineSyncResponse, type MengineUndoState, type MengineUpdateMeta, MirrorVideoDocumentAdapter, type MirrorVideoDocumentOptions, type MoveClipInput, type MoveClipsToStartsInput, type MoveSequentialClipsInput, type MoveVoiceoverInput, type OpActor, type PartAggregation, type PartIdFactory, type PartKind, type PartUnion, type PatchCaptionStyleInput, PlainMemoryAdapter, type PlainMemoryAdapterOptions, type PlannedSemanticOpKind, type PullFailureReason, type PullResult, type PushOutcome, type PushOutcomeKind, type PushResult, type PushResultKind, type ReplaceClipContentInput, type ReplaceMediaClipInput, type ReplaceSequentialClipsInput, type ReplacementMediaClipInput, type ResolvedEntityClipPlacement, type ResolvedEntityTimelineLayout, type ResolvedEntityTrack, SchemaValidator, ScriptCompositionError, type SemanticDocumentAdapter, SemanticEditor, type SemanticOpInput, type SemanticOpKind, type SemanticOpName, type SequentialClipAnchor, type SetBgmInput, type SetCaptionVisibilityInput, type SetClipPlacementInput, type SetClipSpeedInput, type SetClipVolumeInput, type SnapshotReadable, type SolvedVideoDocument, type SpeechHostMap, type SpeechPart, type SpeedShift, TIMELINE_SKELETON_DURATION_MS, type Timeline, type TimelineDoc, type TimelineItem, type Track, type TrackDraft, type TrackItem, type TrackItemDraft, type TrackItemTimePosition, type TransactAudit, type TrimClipInput, type UpdateClipInput, type UpdateClipMarkerInput, ValidationError, type VideoClipPart, type VideoDocument, type VideoDocumentDraft, type VideoDocumentMirrorSchema, VideoDocumentValidationError, type VideoDocumentValidationIssue, type VideoDocumentValidationIssueCode, type VideoDraft, type CaptionPart$1 as VideoDraftCaptionPart, type VideoDraftContent, type VideoDraftPartUnion, type Timeline$1 as VideoDraftTimeline, type Track$1 as VideoDraftTrack, type TrackItem$1 as VideoDraftTrackItem, type VideoMediaAssetFact, type VisualMediaAssetFact, type VoiceoverCaptionFact, type VoiceoverMediaAssetFact, type VoiceoverTakeInput, type VoiceoverTakeResult, type WaitForServerAckOptions, applyEntityRows, applyFieldChanges, arrangeMainTrackSeamlessly, assembleCaptionContent, assembleEntityContent, assemblePhoneticScriptContent, assembleScriptText, assertCanonicalEditorResources, assertEntityProjectionPreservesLegacy, assertFieldChanges, assertMediaAssetWritePolicy, assertValidVideoDocument, audioScriptAssetContent, audioScriptAssetFields, audioScriptAssetHash, base64ToBytes, buildInitialVideoDocument, buildSpeechHostMap, bytesToBase64, cascadeAfterVideoClipChanges, compileEntityRows, createEditSandbox, createMirrorVideoDocument, createMirrorVideoDocumentAdapter, createPlainMemoryAdapter, decodeDocVersionMark, derivePositionFromAbs, effectiveVideoClipDurationMs, encodeDocVersionMark, ensureEditorFoundation, ensureLaneTrack, fillMainTrackTimeGaps, findComposedAudioScript, findLaneTrack, fromVideoDocument, generatePartId, getAt, hasAudioScriptAsset, hostForAbsMs, importMediaAsset, isEmptyVideoClip, isImplementedSemanticOpKind, isMap, laneTrackId, mainTrackRanges, migrateLegacyTimelineToEntities, partDurationMs, partUnionSchema, prepareCaptionContent, projectEntityTimeline, readDocumentAudioScript, readEntityDocument, readMainTrackItems, readMengineEventStream, readPart, readPartDurationMs, readVideoDocumentFromDraft, reassignSpeechesToVideoClipsByTime, recalculateTimelineDuration, relativePositionForAbs, replayJournal, resolveAllSpeechOverlaps, resolveEntityTimelineLayout, resolveFieldPath, resolveSpeechOverlapByShiftingVideos, safeDurationMs, index_d_exports as schemas, selectAudioScriptSegment, snapshotToPlain, solveVideoDocument, speedOf, syncAggregatedClipsTimePosition, toVideoDocument, validateVideoDocument, videoDocumentMirrorSchema, videoDocumentSchema };
package/dist/index.js CHANGED
@@ -1,4 +1,4 @@
1
- import { $ as findLaneTrack, A as assemblePhoneticScriptContent, B as isMediaAssetVariantKind, C as EntityProjectionGraph, D as ScriptCompositionError, E as decodeEntityRelationRows, F as variantBaseEntityIds, G as toVideoDocument, H as buildSpeechHostMap, I as isJsonObject, J as validateVideoDocument, K as VideoDocumentValidationError, L as createEntityId, M as findComposedAudioScript, N as selectAudioScriptSegment, O as assembleCaptionContent, P as updateEntityFields, Q as ensureLaneTrack, R as createRelationId, S as projectEntityTimeline, St as bytesToBase64, T as EntityTimelineProjectionError, U as derivePositionFromAbs, V as buildInitialVideoDocument, W as fromVideoDocument, X as videoDocumentSchema, Y as partUnionSchema, Z as LANE_KINDS_IN_STACK_ORDER, _ as hasAudioScriptAsset, _t as speedOf, a as createMirrorVideoDocument, at as fillMainTrackTimeGaps, b as assertFieldChanges, bt as videoDocumentMirrorSchema, c as DocumentMutationGuard, ct as resolveAllSpeechOverlaps, d as LoroEntityDraft, dt as TIMELINE_SKELETON_DURATION_MS, et as hasResolvedPlacement, f as readEntityDocument, ft as isEmptyVideoClip, g as audioScriptAssetHash, gt as effectiveVideoClipDurationMs, h as audioScriptAssetFields, ht as DEFAULT_UNIT_TIME_MS, i as MirrorVideoDocumentAdapter, it as arrangeMainTrackSeamlessly, j as assembleScriptText, k as assembleEntityContent, l as LORO_ENTITY_SCHEMA, lt as resolveSpeechOverlapByShiftingVideos, m as audioScriptAssetContent, mt as safeDurationMs, n as createPlainMemoryAdapter, nt as solveVideoDocument, o as createMirrorVideoDocumentAdapter, ot as reassignSpeechesToVideoClipsByTime, p as entityOrderGroups, pt as partDurationMs, q as assertValidVideoDocument, r as generatePartId, rt as cascadeAfterVideoClipChanges, s as readVideoDocumentFromDraft, st as recalculateTimelineDuration, t as PlainMemoryAdapter, tt as laneTrackId, u as LoroEntityDocument, ut as syncAggregatedClipsTimePosition, v as equalJson, vt as partUnionToDraft, w as resolveEntityTimelineLayout, x as resolveFieldPath, xt as base64ToBytes, y as applyFieldChanges, yt as recordEntries, z as hasSequence } from "./document-Bnxm1p5E.js";
1
+ import { $ as findLaneTrack, A as assemblePhoneticScriptContent, B as isMediaAssetVariantKind, C as EntityProjectionGraph, D as ScriptCompositionError, E as decodeEntityRelationRows, F as variantBaseEntityIds, G as toVideoDocument, H as buildSpeechHostMap, I as isJsonObject, J as validateVideoDocument, K as VideoDocumentValidationError, L as createEntityId, M as findComposedAudioScript, N as selectAudioScriptSegment, O as assembleCaptionContent, P as updateEntityFields, Q as ensureLaneTrack, R as createRelationId, S as projectEntityTimeline, St as bytesToBase64, T as EntityTimelineProjectionError, U as derivePositionFromAbs, V as buildInitialVideoDocument, W as fromVideoDocument, X as videoDocumentSchema, Y as partUnionSchema, Z as LANE_KINDS_IN_STACK_ORDER, _ as hasAudioScriptAsset, _t as speedOf, a as createMirrorVideoDocument, at as fillMainTrackTimeGaps, b as assertFieldChanges, bt as videoDocumentMirrorSchema, c as DocumentMutationGuard, ct as resolveAllSpeechOverlaps, d as LoroEntityDraft, dt as TIMELINE_SKELETON_DURATION_MS, et as hasResolvedPlacement, f as readEntityDocument, ft as isEmptyVideoClip, g as audioScriptAssetHash, gt as effectiveVideoClipDurationMs, h as audioScriptAssetFields, ht as DEFAULT_UNIT_TIME_MS, i as MirrorVideoDocumentAdapter, it as arrangeMainTrackSeamlessly, j as assembleScriptText, k as assembleEntityContent, l as LORO_ENTITY_SCHEMA, lt as resolveSpeechOverlapByShiftingVideos, m as audioScriptAssetContent, mt as safeDurationMs, n as createPlainMemoryAdapter, nt as solveVideoDocument, o as createMirrorVideoDocumentAdapter, ot as reassignSpeechesToVideoClipsByTime, p as entityOrderGroups, pt as partDurationMs, q as assertValidVideoDocument, r as generatePartId, rt as cascadeAfterVideoClipChanges, s as readVideoDocumentFromDraft, st as recalculateTimelineDuration, t as PlainMemoryAdapter, tt as laneTrackId, u as LoroEntityDocument, ut as syncAggregatedClipsTimePosition, v as equalJson, vt as partUnionToDraft, w as resolveEntityTimelineLayout, x as resolveFieldPath, xt as base64ToBytes, y as applyFieldChanges, yt as recordEntries, z as hasSequence } from "./document-DdwzfZ1g.js";
2
2
  import { addSpeechesInputSchema, addVideoClipsInputSchema, adjustBgmVolumeInputSchema, adjustSpeechVolumeInputSchema, adjustVideoClipDurationInputSchema, adjustVideoClipVolumeInputSchema, changeSpeechScriptInputSchema, changeSpeechVoiceInputSchema, deleteBgmInputSchema, deleteSpeechesInputSchema, deleteVideoClipsInputSchema, moveSpeechesInputSchema, moveVideoClipsByAnchorInputSchema, moveVideoClipsInputSchema, replaceVideoClipContentInputSchema, replaceVideoClipSequenceInputSchema, setBgmInputSchema, setCaptionStyleInputSchema, setCaptionVisibilityInputSchema, setVideoClipSpeedShiftInputSchema, t as schemas_exports } from "./schemas.js";
3
3
  import { LoroDoc, UndoManager, VersionVector } from "loro-crdt";
4
4
  import { ClientServerSynchronizer, DocManager } from "@mengine/sync";
@@ -579,36 +579,55 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
579
579
  "image",
580
580
  "video",
581
581
  "audio",
582
- "voice"
582
+ "voiceover"
583
583
  ].includes(fact.kind)) throw new Error("Unsupported media kind");
584
584
  if (fact.kind !== "image" && (!Number.isFinite(fact.durationMs) || fact.durationMs <= 0)) throw new Error(`${fact.kind} ${fact.assetId} requires its factual positive duration`);
585
- if ((fact.kind === "audio" || fact.kind === "voice") && fact.storageKey.trim() === "") throw new Error(`${fact.kind} ${fact.assetId} requires its physical storage key`);
585
+ if ((fact.kind === "audio" || fact.kind === "voiceover") && fact.storageKey.trim() === "") throw new Error(`${fact.kind} ${fact.assetId} requires its physical storage key`);
586
586
  decodeEntityRelationRows(initial, numericCoordinates);
587
587
  assertCanonicalEditorResources(initial);
588
588
  const entities = structuredClone([...initial.entities]);
589
589
  const relations = structuredClone([...initial.relations]);
590
- const system = fact.kind === "voice" ? "memota-speech" : "memota";
590
+ const system = fact.kind === "voiceover" ? "memota-speech" : "memota";
591
+ const entityKind = fact.kind === "voiceover" ? "audio" : fact.kind;
592
+ const timbre = fact.kind === "voiceover" ? fact.voice : void 0;
591
593
  const existing = entities.filter((row) => {
592
594
  const external = row.payload.external;
593
595
  return isMediaAssetVariantKind(row.entityKind) && isJsonObject(external) && external.system === system && external.key === fact.assetId;
594
596
  });
597
+ const requireMatchingVariant = (row) => {
598
+ const extent = row.payload.extent;
599
+ if (row.entityKind !== entityKind || fact.kind !== "image" && (!isJsonObject(extent) || extent.start !== 0 || extent.end !== fact.durationMs)) throw new Error(`Media facts conflict with existing identity for ${fact.assetId}`);
600
+ if (fact.storageKey !== void 0 && row.payload.storageKey !== fact.storageKey) {
601
+ if (row.payload.storageKey !== void 0) throw new Error(`Media facts conflict with existing storage key for ${fact.assetId}`);
602
+ return {
603
+ ...row,
604
+ payload: {
605
+ ...row.payload,
606
+ storageKey: fact.storageKey
607
+ }
608
+ };
609
+ }
610
+ return row;
611
+ };
595
612
  if (existing.length > 1) throw new Error(`Duplicate media identities for ${fact.assetId}; reuse ${existing[0].entityId}`);
596
613
  if (existing[0] !== void 0) {
597
- const variant = requireMatchingVariant(existing[0], fact);
614
+ const variant = requireMatchingVariant(existing[0]);
598
615
  replaceEntity$1(entities, variant.entityId, variant);
616
+ const voiceEntityId = bindVoiceTimbre(entities, relations, variant.entityId, timbre, idFactory);
599
617
  return {
600
618
  rows: {
601
619
  entities,
602
620
  relations
603
621
  },
604
- contentEntityId: variant.entityId
622
+ contentEntityId: variant.entityId,
623
+ ...voiceEntityId
605
624
  };
606
625
  }
607
626
  const entityId = createEntityId(idFactory("entity"));
608
627
  if (entities.some((row) => row.entityId === entityId)) throw new Error(`Duplicate entity id ${entityId}`);
609
628
  const variant = {
610
629
  entityId,
611
- entityKind: fact.kind,
630
+ entityKind,
612
631
  payload: {
613
632
  extent: fact.kind === "image" ? {
614
633
  kind: "unbounded",
@@ -624,15 +643,11 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
624
643
  system,
625
644
  key: fact.assetId
626
645
  },
627
- ...fact.storageKey === void 0 ? {} : { storageKey: fact.storageKey },
628
- ...fact.kind === "voice" && fact.voice !== void 0 ? { voice: {
629
- system: fact.voice.system,
630
- key: fact.voice.key,
631
- ...fact.voice.name === void 0 ? {} : { name: fact.voice.name }
632
- } } : {}
646
+ ...fact.storageKey === void 0 ? {} : { storageKey: fact.storageKey }
633
647
  }
634
648
  };
635
649
  entities.push(variant);
650
+ const voiceEntityId = bindVoiceTimbre(entities, relations, entityId, timbre, idFactory);
636
651
  const rows = {
637
652
  entities,
638
653
  relations
@@ -641,33 +656,58 @@ function importMediaAsset(initial, fact, idFactory = (prefix) => `${prefix}_${gl
641
656
  assertCanonicalEditorResources(rows);
642
657
  return {
643
658
  rows,
644
- contentEntityId: entityId
659
+ contentEntityId: entityId,
660
+ ...voiceEntityId
645
661
  };
646
662
  }
647
- function requireMatchingVariant(row, fact) {
648
- const extent = row.payload.extent;
649
- if (row.entityKind !== fact.kind || fact.kind !== "image" && (!isJsonObject(extent) || extent.start !== 0 || extent.end !== fact.durationMs) || fact.kind === "voice" && !sameVoice(row.payload.voice, fact.voice)) throw new Error(`Media facts conflict with existing identity for ${fact.assetId}`);
650
- if (fact.storageKey !== void 0 && row.payload.storageKey !== fact.storageKey) {
651
- if (row.payload.storageKey !== void 0) throw new Error(`Media facts conflict with existing storage key for ${fact.assetId}`);
652
- return {
653
- ...row,
654
- payload: {
655
- ...row.payload,
656
- storageKey: fact.storageKey
657
- }
658
- };
663
+ /**
664
+ * Attach the timbre a synthesized voiceover was rendered with. One Voice
665
+ * identity is shared by every take naming the same library entry, so an
666
+ * existing row is reused rather than duplicated. Recorded media declares no
667
+ * timbre and keeps no link.
668
+ */
669
+ function bindVoiceTimbre(entities, relations, audioEntityId, descriptor, idFactory) {
670
+ const linked = relations.filter((relation) => relation.relationKind === "voice-timbre" && (relation.endpoint0EntityId === audioEntityId || relation.endpoint1EntityId === audioEntityId));
671
+ if (linked.length > 1) throw new Error(`Voiceover ${audioEntityId} has multiple Voice timbre relations`);
672
+ if (descriptor === void 0) {
673
+ if (linked.length > 0) throw new Error(`Recorded media ${audioEntityId} cannot keep a synthesized Voice timbre`);
674
+ return {};
659
675
  }
660
- return row;
676
+ const voice = entities.find((row) => row.entityKind === "voice" && sameVoiceDescriptor(row.payload.voice, descriptor)) ?? appendVoiceIdentity(entities, descriptor, idFactory);
677
+ const existing = linked[0];
678
+ if (existing === void 0) relations.push({
679
+ relationId: createRelationId(idFactory("relation")),
680
+ relationKind: "voice-timbre",
681
+ endpoint0EntityId: createEntityId(audioEntityId),
682
+ endpoint1EntityId: createEntityId(voice.entityId),
683
+ metadata: {},
684
+ trace: {}
685
+ });
686
+ else if (existing.endpoint0EntityId !== audioEntityId || existing.endpoint1EntityId !== voice.entityId) throw new Error(`Voiceover ${audioEntityId} already carries a different Voice timbre`);
687
+ return { voiceEntityId: voice.entityId };
688
+ }
689
+ function appendVoiceIdentity(entities, descriptor, idFactory) {
690
+ const voice = {
691
+ entityId: createEntityId(idFactory("entity")),
692
+ entityKind: "voice",
693
+ payload: { voice: {
694
+ system: descriptor.system,
695
+ key: descriptor.key,
696
+ ...descriptor.name === void 0 ? {} : { name: descriptor.name }
697
+ } }
698
+ };
699
+ entities.push(voice);
700
+ return voice;
701
+ }
702
+ /** Two takes share a Voice identity only when they name the same library entry. */
703
+ function sameVoiceDescriptor(stored, descriptor) {
704
+ return isJsonObject(stored) && stored.system === descriptor.system && stored.key === descriptor.key && stored.name === descriptor.name;
661
705
  }
662
706
  function replaceEntity$1(entities, entityId, replacement) {
663
707
  const index = entities.findIndex((row) => row.entityId === entityId);
664
708
  if (index < 0) throw new Error(`Entity id ${entityId} does not exist`);
665
709
  entities[index] = replacement;
666
710
  }
667
- function sameVoice(left, right) {
668
- if (right === void 0) return left === void 0;
669
- return isJsonObject(left) && left.system === right.system && left.key === right.key && left.name === right.name && Object.keys(left).every((key) => key === "system" || key === "key" || key === "name");
670
- }
671
711
  //#endregion
672
712
  //#region src/entity-editor/entity-timeline-editor.ts
673
713
  /**
@@ -1203,20 +1243,20 @@ var EntityTimelineEditor = class {
1203
1243
  hostClipEntityId: input.hostClipEntityId,
1204
1244
  anchorOffset: input.anchorOffset
1205
1245
  };
1206
- if (input.placement !== void 0 && (input.hostClipEntityId !== void 0 || input.anchorOffset !== void 0)) throw new Error("Specify Voice placement or hostClipEntityId/anchorOffset, not both");
1246
+ if (input.placement !== void 0 && (input.hostClipEntityId !== void 0 || input.anchorOffset !== void 0)) throw new Error("Specify voiceover placement or hostClipEntityId/anchorOffset, not both");
1207
1247
  requireVolume(input.volume);
1208
1248
  const imported = importMediaAsset(draft, input.media, this.idFactory);
1209
1249
  replaceRows(draft, imported.rows);
1210
- const voice = requireEntityKind(draft, imported.contentEntityId, "voice");
1250
+ const voiceover = requireEntityKind(draft, imported.contentEntityId, "audio");
1211
1251
  const phonetic = this.requireComposedPhoneticScript(draft, input.phoneticScriptEntityId);
1212
- this.ensurePhoneticRender(draft, voice.entityId, phonetic.entityId);
1252
+ this.ensurePhoneticRender(draft, voiceover.entityId, phonetic.entityId);
1213
1253
  const speechTrack = this.ensureRoleTrack(draft, input.timelineEntityId, "speech");
1214
1254
  const existing = draft.entities.find((row) => row.entityId === input.voiceoverClipEntityId);
1215
1255
  let voiceoverClipEntityId;
1216
1256
  if (existing === void 0) voiceoverClipEntityId = this.insertPlacedClipIntoDraft(draft, {
1217
1257
  clipEntityId: input.voiceoverClipEntityId,
1218
1258
  trackEntityId: speechTrack.entityId,
1219
- contentEntityId: voice.entityId,
1259
+ contentEntityId: voiceover.entityId,
1220
1260
  sourceRange: {
1221
1261
  start: 0,
1222
1262
  end: input.media.durationMs
@@ -1240,7 +1280,7 @@ var EntityTimelineEditor = class {
1240
1280
  });
1241
1281
  this.relinkRelation(draft, topology.markerContentRelation, {
1242
1282
  endpoint0EntityId: topology.marker.entityId,
1243
- endpoint1EntityId: voice.entityId
1283
+ endpoint1EntityId: voiceover.entityId
1244
1284
  });
1245
1285
  this.applyClipPlacement(draft, topology.clip.entityId, placement);
1246
1286
  const placed = requireClipTopology(draft, topology.clip.entityId);
@@ -1259,11 +1299,12 @@ var EntityTimelineEditor = class {
1259
1299
  }
1260
1300
  });
1261
1301
  }
1262
- this.reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voice.entityId, phonetic.entityId, input.captions);
1302
+ this.reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voiceover.entityId, phonetic.entityId, input.captions);
1263
1303
  this.resolveOneVoiceoverOverlap(draft);
1264
1304
  return {
1265
1305
  voiceoverClipEntityId,
1266
- voiceEntityId: voice.entityId,
1306
+ voiceoverAudioEntityId: voiceover.entityId,
1307
+ ...imported.voiceEntityId === void 0 ? {} : { voiceEntityId: imported.voiceEntityId },
1267
1308
  phoneticScriptEntityId: phonetic.entityId,
1268
1309
  audioScriptEntityId: requireComposedAudioScript(draft, phonetic.entityId),
1269
1310
  captionClipEntityIds: input.captions.map((caption) => caption.captionClipEntityId)
@@ -1469,23 +1510,23 @@ var EntityTimelineEditor = class {
1469
1510
  draft.relations.push(this.createRelation(draft, "timeline-track", timeline.entityId, track.entityId));
1470
1511
  return track;
1471
1512
  }
1472
- /** The existing pronunciation variant a Voice consumes; composition is required. */
1513
+ /** The existing pronunciation variant a voiceover consumes; composition is required. */
1473
1514
  requireComposedPhoneticScript(draft, phoneticScriptEntityId) {
1474
1515
  const phonetic = requireEntityKind(draft, phoneticScriptEntityId, "phonetic-script");
1475
1516
  requireComposedAudioScript(draft, phonetic.entityId);
1476
1517
  return phonetic;
1477
1518
  }
1478
1519
  /** Preserve the factual synthesis input, regardless of persisted endpoint positions. */
1479
- ensurePhoneticRender(draft, voiceEntityId, phoneticScriptEntityId) {
1480
- const renders = draft.relations.filter((relation) => relation.relationKind === "phonetic-script-render" && (relation.endpoint0EntityId === voiceEntityId || relation.endpoint1EntityId === voiceEntityId));
1481
- if (renders.length > 1) throw new Error(`Voice "${voiceEntityId}" has multiple PhoneticScript render relations`);
1520
+ ensurePhoneticRender(draft, voiceoverAudioEntityId, phoneticScriptEntityId) {
1521
+ const renders = draft.relations.filter((relation) => relation.relationKind === "phonetic-script-render" && (relation.endpoint0EntityId === voiceoverAudioEntityId || relation.endpoint1EntityId === voiceoverAudioEntityId));
1522
+ if (renders.length > 1) throw new Error(`Voiceover "${voiceoverAudioEntityId}" has multiple PhoneticScript render relations`);
1482
1523
  if (renders[0] !== void 0) {
1483
- if (otherEndpoint(renders[0], voiceEntityId) === phoneticScriptEntityId) return;
1484
- throw new Error(`Voice "${voiceEntityId}" already has a different PhoneticScript synthesis input`);
1524
+ if (otherEndpoint(renders[0], voiceoverAudioEntityId) === phoneticScriptEntityId) return;
1525
+ throw new Error(`Voiceover "${voiceoverAudioEntityId}" already has a different PhoneticScript synthesis input`);
1485
1526
  }
1486
- draft.relations.push(this.createRelation(draft, "phonetic-script-render", voiceEntityId, phoneticScriptEntityId));
1527
+ draft.relations.push(this.createRelation(draft, "phonetic-script-render", voiceoverAudioEntityId, phoneticScriptEntityId));
1487
1528
  }
1488
- reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voiceEntityId, phoneticScriptEntityId, captions) {
1529
+ reconcileVoiceoverCaptions(draft, voiceoverClipEntityId, voiceoverAudioEntityId, phoneticScriptEntityId, captions) {
1489
1530
  const audioScriptEntityId = requireComposedAudioScript(draft, phoneticScriptEntityId);
1490
1531
  const requested = new Set(captions.map((caption) => caption.captionClipEntityId));
1491
1532
  if (requested.size !== captions.length) throw new Error("Caption Clip ids must be unique within a voiceover take");
@@ -1516,7 +1557,7 @@ var EntityTimelineEditor = class {
1516
1557
  };
1517
1558
  draft.entities.push(content);
1518
1559
  assertCaptionScript(draft, content.entityId, audioScriptEntityId);
1519
- draft.relations.push(this.createRelation(draft, "caption-alignment", content.entityId, voiceEntityId, { alignment: {
1560
+ draft.relations.push(this.createRelation(draft, "caption-alignment", content.entityId, voiceoverAudioEntityId, { alignment: {
1520
1561
  start: startMs,
1521
1562
  end: startMs + durationMs
1522
1563
  } }));
@@ -1573,7 +1614,7 @@ var EntityTimelineEditor = class {
1573
1614
  }
1574
1615
  }
1575
1616
  });
1576
- this.replaceCaptionAlignment(draft, content.entityId, voiceEntityId, startMs, durationMs);
1617
+ this.replaceCaptionAlignment(draft, content.entityId, voiceoverAudioEntityId, startMs, durationMs);
1577
1618
  }
1578
1619
  }
1579
1620
  ensureRoleTrackForClip(draft, clipEntityId, role) {
@@ -1582,10 +1623,10 @@ var EntityTimelineEditor = class {
1582
1623
  const timeline = requireEntityKind(draft, otherEndpoint(requireOneRelation(draft, track.entityId, "timeline-track"), track.entityId), "timeline");
1583
1624
  return this.ensureRoleTrack(draft, timeline.entityId, role);
1584
1625
  }
1585
- replaceCaptionAlignment(draft, captionEntityId, voiceEntityId, startMs, durationMs) {
1626
+ replaceCaptionAlignment(draft, captionEntityId, voiceoverAudioEntityId, startMs, durationMs) {
1586
1627
  const removed = new Set(incidentRelations(draft, captionEntityId).filter((relation) => relation.relationKind === "caption-alignment").map((relation) => relation.relationId));
1587
1628
  draft.relations = draft.relations.filter((relation) => !removed.has(relation.relationId));
1588
- draft.relations.push(this.createRelation(draft, "caption-alignment", captionEntityId, voiceEntityId, { alignment: {
1629
+ draft.relations.push(this.createRelation(draft, "caption-alignment", captionEntityId, voiceoverAudioEntityId, { alignment: {
1589
1630
  start: startMs,
1590
1631
  end: startMs + durationMs
1591
1632
  } }));
@@ -2048,7 +2089,7 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
2048
2089
  relations
2049
2090
  }, {
2050
2091
  assetId: speech.origin_speech_id,
2051
- kind: "voice",
2092
+ kind: "voiceover",
2052
2093
  durationMs: speech.media_duration_ms,
2053
2094
  storageKey: speech.audio_storage_key,
2054
2095
  voice: {
@@ -2136,8 +2177,8 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
2136
2177
  } else if (part.caption !== void 0) {
2137
2178
  const caption = part.caption;
2138
2179
  duration = caption.initial_duration_ms;
2139
- const voice = contentByPartId.get(caption.speech_part_id);
2140
- const render = relations.find((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voice);
2180
+ const voiceover = contentByPartId.get(caption.speech_part_id);
2181
+ const render = relations.find((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voiceover);
2141
2182
  const phonetic = entities.find((entity) => entity.entityId === render?.endpoint1EntityId);
2142
2183
  const script = phonetic === void 0 ? void 0 : composedAudioScript(phonetic, entities);
2143
2184
  if (script === void 0) fail("Legacy Caption has no composed AudioScript");
@@ -2164,7 +2205,7 @@ function migrateLegacyTimelineToEntities(legacy, assetFacts, initial = {
2164
2205
  selection,
2165
2206
  ...caption.style === void 0 ? {} : { style: toEntityCaptionStyle(caption.style) }
2166
2207
  });
2167
- link("caption-alignment", content, voice, { alignment: {
2208
+ link("caption-alignment", content, voiceover, { alignment: {
2168
2209
  start: caption.start_ms,
2169
2210
  end: caption.start_ms + duration
2170
2211
  } });
@@ -2503,21 +2544,21 @@ function requirePart(document, partId) {
2503
2544
  if (part === void 0) fail(`Placed Clip ${partId} has no part payload`);
2504
2545
  return part;
2505
2546
  }
2506
- function ensureSpeechPhoneticChain(voiceEntityId, audioScriptText, phonemeScript, entities, relations, addEntity, link, documentAudioScriptEntityId) {
2507
- const renders = relations.filter((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voiceEntityId);
2508
- if (renders.length > 1) fail(`Voice ${voiceEntityId} has ambiguous PhoneticScript provenance`);
2547
+ function ensureSpeechPhoneticChain(voiceoverAudioEntityId, audioScriptText, phonemeScript, entities, relations, addEntity, link, documentAudioScriptEntityId) {
2548
+ const renders = relations.filter((relation) => relation.relationKind === "phonetic-script-render" && relation.endpoint0EntityId === voiceoverAudioEntityId);
2549
+ if (renders.length > 1) fail(`Voiceover ${voiceoverAudioEntityId} has ambiguous PhoneticScript provenance`);
2509
2550
  if (renders.length === 1) {
2510
2551
  const phonetic = entities.find((entity) => entity.entityId === renders[0].endpoint1EntityId && entity.entityKind === "phonetic-script");
2511
- if (phonetic === void 0) fail(`Voice ${voiceEntityId} has a dangling PhoneticScript render`);
2552
+ if (phonetic === void 0) fail(`Voiceover ${voiceoverAudioEntityId} has a dangling PhoneticScript render`);
2512
2553
  const script = composedAudioScript(phonetic, entities);
2513
- if (script === void 0 || scriptText(script) !== audioScriptText) fail(`Voice ${voiceEntityId} conflicts with the legacy Speech script`);
2554
+ if (script === void 0 || scriptText(script) !== audioScriptText) fail(`Voiceover ${voiceoverAudioEntityId} conflicts with the legacy Speech script`);
2514
2555
  return phonetic.entityId;
2515
2556
  }
2516
2557
  const script = entities.find((entity) => entity.entityId === documentAudioScriptEntityId);
2517
2558
  if (script === void 0 || script.entityKind !== "audio-script") fail("Migration must carry the fixed document AudioScript");
2518
2559
  const segments = script.payload.segments;
2519
2560
  if (Array.isArray(segments) && segments.length > 0) {
2520
- if (scriptText(script) !== audioScriptText) fail(`Voice ${voiceEntityId} conflicts with the project AudioScript text`);
2561
+ if (scriptText(script) !== audioScriptText) fail(`Voiceover ${voiceoverAudioEntityId} conflicts with the project AudioScript text`);
2521
2562
  } else {
2522
2563
  const scriptIndex = entities.findIndex((row) => row.entityId === script.entityId);
2523
2564
  entities[scriptIndex] = {
@@ -2532,7 +2573,7 @@ function ensureSpeechPhoneticChain(voiceEntityId, audioScriptText, phonemeScript
2532
2573
  baseEntityIds: [script.entityId],
2533
2574
  ...phonemeScript === void 0 ? {} : { phonemeScript }
2534
2575
  });
2535
- link("phonetic-script-render", voiceEntityId, phonetic);
2576
+ link("phonetic-script-render", voiceoverAudioEntityId, phonetic);
2536
2577
  return phonetic;
2537
2578
  }
2538
2579
  function composedAudioScript(phonetic, entities) {
package/dist/testing.js CHANGED
@@ -1,4 +1,4 @@
1
- import { St as bytesToBase64, o as createMirrorVideoDocumentAdapter, xt as base64ToBytes } from "./document-Bnxm1p5E.js";
1
+ import { St as bytesToBase64, o as createMirrorVideoDocumentAdapter, xt as base64ToBytes } from "./document-DdwzfZ1g.js";
2
2
  import { t as classifyUpdate } from "./loro-relay-doc-cJSY-uau.js";
3
3
  import { LoroDoc, VersionVector, encodeFrontiers } from "loro-crdt";
4
4
  //#region src/testing/in-memory-mengine-server.ts
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@mengine/medeo-client",
3
- "version": "2.0.1-alpha.7",
3
+ "version": "2.0.1-alpha.8",
4
4
  "license": "UNLICENSED",
5
5
  "repository": {
6
6
  "type": "git",
@@ -28,9 +28,9 @@
28
28
  "immer": "^10.2.0",
29
29
  "loro-mirror": "^2.3.3",
30
30
  "zod": "^4.4.3",
31
- "@mengine/storage": "2.0.1-alpha.7",
32
- "@mengine/sync": "2.0.1-alpha.7",
33
- "@mengine/utils": "2.0.1-alpha.7"
31
+ "@mengine/sync": "2.0.1-alpha.8",
32
+ "@mengine/utils": "2.0.1-alpha.8",
33
+ "@mengine/storage": "2.0.1-alpha.8"
34
34
  },
35
35
  "devDependencies": {
36
36
  "@types/node": "^25.9.1",