@nodaro/shared 2.14.0 → 2.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +94 -12
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +145 -11
- package/dist/index.d.ts +145 -11
- package/dist/index.js +90 -13
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
- package/src/__tests__/image-mention-slug.test.ts +303 -0
- package/src/__tests__/prompt-length-limits.test.ts +2 -2
- package/src/__tests__/to-connected-references.test.ts +45 -0
- package/src/__tests__/video-analysis.test.ts +11 -0
- package/src/character-voice.ts +4 -3
- package/src/image-mention-slug.ts +236 -0
- package/src/index.ts +9 -0
- package/src/model-catalog.ts +10 -7
- package/src/model-constants.ts +8 -6
- package/src/model-tree.ts +1 -0
- package/src/to-connected-references.ts +16 -2
- package/src/video-analysis.ts +9 -0
package/src/model-catalog.ts
CHANGED
|
@@ -58,6 +58,10 @@ export type ModelMode =
|
|
|
58
58
|
| "isolation"
|
|
59
59
|
| "dubbing"
|
|
60
60
|
| "forced-alignment"
|
|
61
|
+
// dialogue — deliberately its own mode, never "tts": the dialogue model
|
|
62
|
+
// takes a multi-speaker script shape (inputs[]), not the single-text
|
|
63
|
+
// generate_speech contract, so it must never appear in a TTS model list.
|
|
64
|
+
| "dialogue"
|
|
61
65
|
// video analysis
|
|
62
66
|
| "video-analysis"
|
|
63
67
|
// video audit — deliberately its own mode, never "video-analysis": the
|
|
@@ -2143,18 +2147,17 @@ const AUDIO_MODELS: Record<string, ModelCatalogEntry> = {
|
|
|
2143
2147
|
"elevenlabs-dialogue": {
|
|
2144
2148
|
id: "elevenlabs-dialogue",
|
|
2145
2149
|
kind: "audio",
|
|
2146
|
-
|
|
2150
|
+
// Its own mode, never "tts": the script shape (inputs[]) doesn't fit the
|
|
2151
|
+
// single-text generate_speech contract, so list_models must never offer
|
|
2152
|
+
// it there. The dialogue-capable MCP verb is `generate_dialogue`.
|
|
2153
|
+
modes: ["dialogue"] as const,
|
|
2147
2154
|
family: "ElevenLabs",
|
|
2148
2155
|
label: "ElevenLabs Dialogue v3",
|
|
2149
2156
|
series: "ElevenLabs",
|
|
2150
|
-
description: "Multi-speaker dialogue
|
|
2157
|
+
description: "Multi-speaker dialogue via the direct ElevenLabs API — give it a script, it voices each role (any voice: premade, library, or cloned).",
|
|
2151
2158
|
useCases: ["tts", "dialogue", "multi-speaker"],
|
|
2159
|
+
features: ["audio-tags", "voice-cloning"],
|
|
2152
2160
|
pricing: [{ identifier: "elevenlabs-dialogue", credits: 25, note: "per 1K chars" }],
|
|
2153
|
-
// Driven only via the dialogue/character-voice path (multi-speaker script
|
|
2154
|
-
// shape), NOT the single-text generate_speech verb. Hide from MCP
|
|
2155
|
-
// list_models so generate_speech (TTS_PROVIDERS) can't advertise it and
|
|
2156
|
-
// then 400. Re-expose if a dialogue-capable MCP verb is added.
|
|
2157
|
-
mcpHidden: true,
|
|
2158
2161
|
},
|
|
2159
2162
|
|
|
2160
2163
|
// ── ElevenLabs voice utilities ──
|
package/src/model-constants.ts
CHANGED
|
@@ -211,16 +211,18 @@ export const TTS_TEXT_MAX = 5000
|
|
|
211
211
|
|
|
212
212
|
/**
|
|
213
213
|
* Per-model Text-to-Speech character cap (PER REQUEST), from official ElevenLabs
|
|
214
|
-
* docs. turbo/multilingual accept FAR more than the old flat 5000; v3
|
|
215
|
-
*
|
|
216
|
-
* hard-
|
|
217
|
-
*
|
|
214
|
+
* docs. turbo/multilingual accept FAR more than the old flat 5000; v3 matches
|
|
215
|
+
* the official 5000 (probed live 2026-08-30: 4,500 AND 5,200 chars both
|
|
216
|
+
* returned 200 — the old "API hard-limits v3 at 3000" report no longer holds).
|
|
217
|
+
* Dialogue's documented 2,000 is a recommendation, not a limit (2,500 and
|
|
218
|
+
* 5,000 total chars both probed 200); we cap at 5000 like v3 — same model
|
|
219
|
+
* underneath. Absent → {@link TTS_TEXT_MAX}.
|
|
218
220
|
*/
|
|
219
221
|
export const MAX_TTS_CHARS_BY_PROVIDER: Record<string, number> = {
|
|
220
222
|
"elevenlabs-turbo": 40000, // == eleven_flash_v2_5 (functionally equivalent)
|
|
221
223
|
"elevenlabs-multilingual": 10000, // eleven_multilingual_v2
|
|
222
|
-
"elevenlabs-v3":
|
|
223
|
-
"elevenlabs-dialogue":
|
|
224
|
+
"elevenlabs-v3": 5000, // official cap (probed: 5,200 chars accepted; keep the clamp)
|
|
225
|
+
"elevenlabs-dialogue": 5000, // total across lines; ≤2,000 recommended for best quality
|
|
224
226
|
}
|
|
225
227
|
|
|
226
228
|
/** Max TTS text length (chars) for a provider: verified override, else {@link TTS_TEXT_MAX}. */
|
package/src/model-tree.ts
CHANGED
|
@@ -62,6 +62,7 @@ const MODE_TO_NODE: Readonly<Partial<Record<ModelMode, string>>> = {
|
|
|
62
62
|
"lip-sync": "lip-sync", "tts": "text-to-speech", "sfx": "text-to-audio",
|
|
63
63
|
"music": "suno-generate", "voice-design": "voice-design", "voice-changer": "voice-changer",
|
|
64
64
|
"isolation": "audio-isolation", "dubbing": "dubbing", "forced-alignment": "forced-alignment", "stt": "transcribe",
|
|
65
|
+
"dialogue": "text-to-dialogue",
|
|
65
66
|
"video-analysis": "video-analysis",
|
|
66
67
|
}
|
|
67
68
|
|
|
@@ -12,7 +12,7 @@ import { locationMentionSlug } from "./location-mention-slug.js"
|
|
|
12
12
|
export interface EntityReferenceInput {
|
|
13
13
|
/** The entity row id. */
|
|
14
14
|
readonly id: string
|
|
15
|
-
readonly kind: "character" | "location" | "creature"
|
|
15
|
+
readonly kind: "character" | "location" | "creature" | "image"
|
|
16
16
|
/** Display name — the slug is derived from this. */
|
|
17
17
|
readonly name: string
|
|
18
18
|
/** Resolved thumbnail/source URL; null/undefined → "" (placeholder-safe). */
|
|
@@ -30,7 +30,9 @@ export interface EntityReferenceInput {
|
|
|
30
30
|
* `locationMentionSlug`), creature → `wired-creature` (no mention-slug machinery
|
|
31
31
|
* — like `wired-object`, a bound creature AUTO-ATTACHES and gets a canonical-style
|
|
32
32
|
* creature/animal-subject directive with zero typing; `{image:N:creature}` tokens
|
|
33
|
-
* also resolve against it)
|
|
33
|
+
* also resolve against it), image → `wired-image` (a plain media reference; it
|
|
34
|
+
* AUTO-ATTACHES, and an `@<name-slug>:<index>[:<role>]` mention re-seats it at the
|
|
35
|
+
* position it was typed). Canonical-description fields are null here (a binding
|
|
34
36
|
* captures only name+variant; the picker/full-row path supplies descriptions).
|
|
35
37
|
*/
|
|
36
38
|
export function toConnectedReference(entity: EntityReferenceInput): ConnectedReference {
|
|
@@ -47,6 +49,18 @@ export function toConnectedReference(entity: EntityReferenceInput): ConnectedRef
|
|
|
47
49
|
variantDisplayName: entity.variant ?? "canonical",
|
|
48
50
|
}
|
|
49
51
|
}
|
|
52
|
+
if (entity.kind === "image") {
|
|
53
|
+
// No slug field — the resolver derives it from `defaultName` at prompt time
|
|
54
|
+
// (`knownImageSlugsFromRefs`), so a client cannot drift from the grammar and
|
|
55
|
+
// nothing changes on the wire.
|
|
56
|
+
return {
|
|
57
|
+
id: entity.id,
|
|
58
|
+
defaultName: entity.name,
|
|
59
|
+
source: "wired-image",
|
|
60
|
+
url: entity.url ?? "",
|
|
61
|
+
...(entity.description ? { description: entity.description } : {}),
|
|
62
|
+
}
|
|
63
|
+
}
|
|
50
64
|
if (entity.kind === "creature") {
|
|
51
65
|
return {
|
|
52
66
|
id: entity.id,
|
package/src/video-analysis.ts
CHANGED
|
@@ -334,6 +334,15 @@ const windowSceneBase = z.object({
|
|
|
334
334
|
/** Effects on this shot's PICTURE. Absent ⇒ a clean image. */
|
|
335
335
|
effects: z.array(z.enum(VIDEO_ANALYSIS_VISUAL_EFFECTS)).optional(),
|
|
336
336
|
transitionOut: z.enum(VIDEO_ANALYSIS_TRANSITIONS).optional(),
|
|
337
|
+
// Round 4 (2026-08-30, recast shot craft rev 1.5 Appendix F.8): the edit into
|
|
338
|
+
// the next scene in the analyser's OWN WORDS — a few words of what it saw
|
|
339
|
+
// ("fast blurred pan left", "fade to black", "zoom-blur warp into the tunnel")
|
|
340
|
+
// — or absent when the take continues through the change or no edit is seen
|
|
341
|
+
// (nothing asserted: the video model decides from story and refs). The closed
|
|
342
|
+
// `transitionOut` vocabulary above is LEGACY from here on: kept so stored
|
|
343
|
+
// blueprints still parse, never emitted again. Readers take
|
|
344
|
+
// `transition ?? transitionOut`.
|
|
345
|
+
transition: z.string().trim().min(1).max(120).optional(),
|
|
337
346
|
// Array of concurrent layers (music + speech + sfx together); [] = silence.
|
|
338
347
|
audio: z.array(audioLayerSchema),
|
|
339
348
|
/** slotId → variationId for slots wearing a NON-default look in this scene
|