@nebutra/voice-realtime 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md ADDED
@@ -0,0 +1,20 @@
1
+ # @nebutra/voice-realtime
2
+
3
+ Status: WIP — Not yet integrated into any production app.
4
+
5
+ `@nebutra/voice-realtime` owns voice session lifecycle, narration synthesis,
6
+ interrupt metadata, and voice enrollment contracts. The zero-config path writes
7
+ valid narration WAV files through audio utilities; realtime WebRTC/STT/TTS is a
8
+ sidecar port.
9
+
10
+ It does not own Thread/Turn/Item state, model provider routing, or prompt
11
+ orchestration.
12
+
13
+ ## Commands
14
+
15
+ ```bash
16
+ pnpm voice:doctor
17
+ pnpm voice:debug
18
+ pnpm voice:enroll
19
+ pnpm voice:test-mic
20
+ ```
@@ -0,0 +1,108 @@
1
+ // src/index.ts
2
+ import { mkdir, writeFile } from "fs/promises";
3
+ import { dirname, join } from "path";
4
+ import { createToneWav } from "@nebutra/audio-pipeline";
5
+ import { appendCapabilityDebug, readCapabilityDebug } from "@nebutra/capability-kit/debug";
6
+ import {
7
+ assetId,
8
+ requireBrandContext
9
+ } from "@nebutra/generation-context";
10
+ async function readVoiceDebug(root = process.cwd(), limit = 10) {
11
+ return readCapabilityDebug("voice-realtime", { root, limit });
12
+ }
13
+ var VoiceRealtime = class {
14
+ #root;
15
+ constructor(options = {}) {
16
+ this.#root = options.root ?? process.cwd();
17
+ }
18
+ async startSession(input) {
19
+ const session = {
20
+ id: assetId("voice_session", input.threadId),
21
+ tenantId: input.tenantId,
22
+ threadId: input.threadId,
23
+ room: `voice_${input.tenantId}_${input.threadId}`,
24
+ state: "listening"
25
+ };
26
+ await appendCapabilityDebug(
27
+ "voice-realtime",
28
+ { type: "session", session },
29
+ { root: this.#root }
30
+ );
31
+ return session;
32
+ }
33
+ async synthesizeNarration(request, brandInput) {
34
+ const brand = requireBrandContext(brandInput, "voice-realtime");
35
+ const durationS = Math.max(
36
+ 1,
37
+ Math.min(request.targetDurationS ?? Math.ceil(request.script.length / 24), 20)
38
+ );
39
+ const id = assetId("voice", `${brand.brandId}_${request.script}`);
40
+ const path = join(this.#root, ".nebutra", "generated", "voice-realtime", `${id}.wav`);
41
+ await mkdir(dirname(path), { recursive: true });
42
+ await writeFile(path, createToneWav(durationS, 330));
43
+ const asset = {
44
+ id,
45
+ tenantId: brand.tenantId,
46
+ kind: "voice",
47
+ path,
48
+ brandId: brand.brandId,
49
+ provider: "tone-local",
50
+ model: "brand-context-narration-v1",
51
+ createdAt: (/* @__PURE__ */ new Date()).toISOString(),
52
+ license: { status: "commercial-ok", source: "deterministic local renderer" },
53
+ transcript: request.script,
54
+ durationS,
55
+ format: "wav",
56
+ voiceProfileId: request.voiceProfileId ?? "default",
57
+ metadata: { brandSource: brand.sourcePath }
58
+ };
59
+ await appendCapabilityDebug(
60
+ "voice-realtime",
61
+ { type: "narration", asset },
62
+ { root: this.#root }
63
+ );
64
+ return asset;
65
+ }
66
+ async enroll(input) {
67
+ const profile = {
68
+ id: assetId("voice_profile", input.tenantId),
69
+ tenantId: input.tenantId,
70
+ consentRecordedAt: (/* @__PURE__ */ new Date()).toISOString(),
71
+ sampleCount: input.samplePaths?.length ?? 0
72
+ };
73
+ await appendCapabilityDebug(
74
+ "voice-realtime",
75
+ { type: "enroll", profile },
76
+ { root: this.#root }
77
+ );
78
+ return profile;
79
+ }
80
+ async testMic() {
81
+ return {
82
+ provider: "local-mic",
83
+ ok: Boolean(process.env.VOICE_MIC_PERMISSION === "granted"),
84
+ suggestion: "Grant microphone access in the desktop app before realtime sessions."
85
+ };
86
+ }
87
+ async doctor() {
88
+ return [
89
+ await this.testMic(),
90
+ {
91
+ provider: "voice-sidecar",
92
+ ok: Boolean(process.env.VOICE_SIDECAR_URL),
93
+ suggestion: "Set VOICE_SIDECAR_URL to enable realtime WebRTC/STT/TTS."
94
+ },
95
+ {
96
+ provider: "remote-tts",
97
+ ok: Boolean(process.env.VOICE_REMOTE_API_KEY),
98
+ suggestion: "Set VOICE_REMOTE_API_KEY to enable remote voice fallback."
99
+ }
100
+ ];
101
+ }
102
+ };
103
+
104
+ export {
105
+ readVoiceDebug,
106
+ VoiceRealtime
107
+ };
108
+ //# sourceMappingURL=chunk-GARLTI7J.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/index.ts"],"sourcesContent":["import { mkdir, writeFile } from \"node:fs/promises\";\nimport { dirname, join } from \"node:path\";\nimport { createToneWav } from \"@nebutra/audio-pipeline\";\nimport { appendCapabilityDebug, readCapabilityDebug } from \"@nebutra/capability-kit/debug\";\nimport {\n assetId,\n type BrandContext,\n type GeneratedAsset,\n requireBrandContext,\n} from \"@nebutra/generation-context\";\n\nexport type VoiceState = \"listening\" | \"thinking\" | \"speaking\" | \"paused\" | \"closed\";\n\nexport interface VoiceSession {\n readonly id: string;\n readonly tenantId: string;\n readonly threadId: string;\n readonly room: string;\n readonly state: VoiceState;\n}\n\nexport interface NarrationRequest {\n readonly script: string;\n readonly voiceProfileId?: string;\n readonly targetDurationS?: number;\n}\n\nexport interface VoiceAsset extends GeneratedAsset {\n readonly kind: \"voice\";\n readonly transcript: string;\n readonly durationS: number;\n readonly format: \"wav\";\n readonly voiceProfileId: string;\n}\n\nexport interface VoiceProfile {\n readonly id: string;\n readonly tenantId: string;\n readonly consentRecordedAt: string;\n readonly sampleCount: number;\n}\n\nexport interface VoiceHealth {\n readonly provider: string;\n readonly ok: boolean;\n readonly suggestion?: string;\n}\n\nexport interface VoiceRealtimeOptions {\n readonly root?: string;\n}\n\nexport async function readVoiceDebug(root = process.cwd(), limit = 10): Promise<unknown[]> {\n return readCapabilityDebug(\"voice-realtime\", { root, limit });\n}\n\nexport class VoiceRealtime {\n readonly #root: string;\n\n constructor(options: VoiceRealtimeOptions = {}) {\n this.#root = options.root ?? process.cwd();\n }\n\n async startSession(input: { tenantId: string; threadId: string }): Promise<VoiceSession> {\n const session: VoiceSession = {\n id: assetId(\"voice_session\", input.threadId),\n tenantId: input.tenantId,\n threadId: input.threadId,\n room: `voice_${input.tenantId}_${input.threadId}`,\n state: \"listening\",\n };\n await appendCapabilityDebug(\n \"voice-realtime\",\n { type: \"session\", session },\n { root: this.#root },\n );\n return session;\n }\n\n async synthesizeNarration(\n request: NarrationRequest,\n brandInput: BrandContext | undefined,\n ): Promise<VoiceAsset> {\n const brand = requireBrandContext(brandInput, \"voice-realtime\");\n const durationS = Math.max(\n 1,\n Math.min(request.targetDurationS ?? Math.ceil(request.script.length / 24), 20),\n );\n const id = assetId(\"voice\", `${brand.brandId}_${request.script}`);\n const path = join(this.#root, \".nebutra\", \"generated\", \"voice-realtime\", `${id}.wav`);\n await mkdir(dirname(path), { recursive: true });\n await writeFile(path, createToneWav(durationS, 330));\n const asset: VoiceAsset = {\n id,\n tenantId: brand.tenantId,\n kind: \"voice\",\n path,\n brandId: brand.brandId,\n provider: \"tone-local\",\n model: \"brand-context-narration-v1\",\n createdAt: new Date().toISOString(),\n license: { status: \"commercial-ok\", source: \"deterministic local renderer\" },\n transcript: request.script,\n durationS,\n format: \"wav\",\n voiceProfileId: request.voiceProfileId ?? \"default\",\n metadata: { brandSource: brand.sourcePath },\n };\n await appendCapabilityDebug(\n \"voice-realtime\",\n { type: \"narration\", asset },\n { root: this.#root },\n );\n return asset;\n }\n\n async enroll(input: {\n tenantId: string;\n samplePaths?: readonly string[];\n }): Promise<VoiceProfile> {\n const profile: VoiceProfile = {\n id: assetId(\"voice_profile\", input.tenantId),\n tenantId: input.tenantId,\n consentRecordedAt: new Date().toISOString(),\n sampleCount: input.samplePaths?.length ?? 0,\n };\n await appendCapabilityDebug(\n \"voice-realtime\",\n { type: \"enroll\", profile },\n { root: this.#root },\n );\n return profile;\n }\n\n async testMic(): Promise<VoiceHealth> {\n return {\n provider: \"local-mic\",\n ok: Boolean(process.env.VOICE_MIC_PERMISSION === \"granted\"),\n suggestion: \"Grant microphone access in the desktop app before realtime sessions.\",\n };\n }\n\n async doctor(): Promise<VoiceHealth[]> {\n return [\n await this.testMic(),\n {\n provider: \"voice-sidecar\",\n ok: Boolean(process.env.VOICE_SIDECAR_URL),\n suggestion: \"Set VOICE_SIDECAR_URL to enable realtime WebRTC/STT/TTS.\",\n },\n {\n provider: \"remote-tts\",\n ok: Boolean(process.env.VOICE_REMOTE_API_KEY),\n suggestion: \"Set VOICE_REMOTE_API_KEY to enable remote voice fallback.\",\n },\n ];\n }\n}\n"],"mappings":";AAAA,SAAS,OAAO,iBAAiB;AACjC,SAAS,SAAS,YAAY;AAC9B,SAAS,qBAAqB;AAC9B,SAAS,uBAAuB,2BAA2B;AAC3D;AAAA,EACE;AAAA,EAGA;AAAA,OACK;AA2CP,eAAsB,eAAe,OAAO,QAAQ,IAAI,GAAG,QAAQ,IAAwB;AACzF,SAAO,oBAAoB,kBAAkB,EAAE,MAAM,MAAM,CAAC;AAC9D;AAEO,IAAM,gBAAN,MAAoB;AAAA,EAChB;AAAA,EAET,YAAY,UAAgC,CAAC,GAAG;AAC9C,SAAK,QAAQ,QAAQ,QAAQ,QAAQ,IAAI;AAAA,EAC3C;AAAA,EAEA,MAAM,aAAa,OAAsE;AACvF,UAAM,UAAwB;AAAA,MAC5B,IAAI,QAAQ,iBAAiB,MAAM,QAAQ;AAAA,MAC3C,UAAU,MAAM;AAAA,MAChB,UAAU,MAAM;AAAA,MAChB,MAAM,SAAS,MAAM,QAAQ,IAAI,MAAM,QAAQ;AAAA,MAC/C,OAAO;AAAA,IACT;AACA,UAAM;AAAA,MACJ;AAAA,MACA,EAAE,MAAM,WAAW,QAAQ;AAAA,MAC3B,EAAE,MAAM,KAAK,MAAM;AAAA,IACrB;AACA,WAAO;AAAA,EACT;AAAA,EAEA,MAAM,oBACJ,SACA,YACqB;AACrB,UAAM,QAAQ,oBAAoB,YAAY,gBAAgB;AAC9D,UAAM,YAAY,KAAK;AAAA,MACrB;AAAA,MACA,KAAK,IAAI,QAAQ,mBAAmB,KAAK,KAAK,QAAQ,OAAO,SAAS,EAAE,GAAG,EAAE;AAAA,IAC/E;AACA,UAAM,KAAK,QAAQ,SAAS,GAAG,MAAM,OAAO,IAAI,QAAQ,MAAM,EAAE;AAChE,UAAM,OAAO,KAAK,KAAK,OAAO,YAAY,aAAa,kBAAkB,GAAG,EAAE,MAAM;AACpF,UAAM,MAAM,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;AAC9C,UAAM,UAAU,MAAM,cAAc,WAAW,GAAG,CAAC;AACnD,UAAM,QAAoB;AAAA,MACxB;AAAA,MACA,UAAU,MAAM;AAAA,MAChB,MAAM;AAAA,MACN;AAAA,MACA,SAAS,MAAM;AAAA,MACf,UAAU;AAAA,MACV,OAAO;AAAA,MACP,YAAW,oBAAI,KAAK,GAAE,YAAY;AAAA,MAClC,SAAS,EAAE,QAAQ,iBAAiB,QAAQ,+BAA+B;AAAA,MAC3E,YAAY,QAAQ;AAAA,MACpB;AAAA,MACA,QAAQ;AAAA,MACR,gBAAgB,QAAQ,kBAAkB;AAAA,MAC1C,UAAU,EAAE,aAAa,MAAM,WAAW;AAAA,IAC5C;AACA,UAAM;AAAA,MACJ;AAAA,MACA,EAAE,MAAM,aAAa,MAAM;AAAA,MAC3B,EAAE,MAAM,KAAK,MAAM;AAAA,IACrB;AACA,WAAO;AAAA,EACT;AAAA,EAEA,MAAM,OAAO,OAGa;AACxB,UAAM,UAAwB;AAAA,MAC5B,IAAI,QAAQ,iBAAiB,MAAM,QAAQ;AAAA,MAC3C,UAAU,MAAM;AAAA,MAChB,oBAAmB,oBAAI,KAAK,GAAE,YAAY;AAAA,MAC1C,aAAa,MAAM,aAAa,UAAU;AAAA,IAC5C;AACA,UAAM;AAAA,MACJ;AAAA,MACA,EAAE,MAAM,UAAU,QAAQ;AAAA,MAC1B,EAAE,MAAM,KAAK,MAAM;AAAA,IACrB;AACA,WAAO;AAAA,EACT;AAAA,EAEA,MAAM,UAAgC;AACpC,WAAO;AAAA,MACL,UAAU;AAAA,MACV,IAAI,QAAQ,QAAQ,IAAI,yBAAyB,SAAS;AAAA,MAC1D,YAAY;AAAA,IACd;AAAA,EACF;AAAA,EAEA,MAAM,SAAiC;AACrC,WAAO;AAAA,MACL,MAAM,KAAK,QAAQ;AAAA,MACnB;AAAA,QACE,UAAU;AAAA,QACV,IAAI,QAAQ,QAAQ,IAAI,iBAAiB;AAAA,QACzC,YAAY;AAAA,MACd;AAAA,MACA;AAAA,QACE,UAAU;AAAA,QACV,IAAI,QAAQ,QAAQ,IAAI,oBAAoB;AAAA,QAC5C,YAAY;AAAA,MACd;AAAA,IACF;AAAA,EACF;AACF;","names":[]}
package/dist/cli.d.ts ADDED
@@ -0,0 +1,2 @@
1
+
2
+ export { }
package/dist/cli.js ADDED
@@ -0,0 +1,42 @@
1
+ import {
2
+ VoiceRealtime,
3
+ readVoiceDebug
4
+ } from "./chunk-GARLTI7J.js";
5
+
6
+ // src/cli.ts
7
+ import { createDemoBrandContext } from "@nebutra/generation-context";
8
+ var command = process.argv[2] ?? "doctor";
9
+ var voice = new VoiceRealtime();
10
+ if (command === "doctor") {
11
+ process.stdout.write(
12
+ `${JSON.stringify({ capability: "voice-realtime", results: await voice.doctor() }, null, 2)}
13
+ `
14
+ );
15
+ } else if (command === "debug") {
16
+ process.stdout.write(
17
+ `${JSON.stringify({ capability: "voice-realtime", entries: await readVoiceDebug() }, null, 2)}
18
+ `
19
+ );
20
+ } else if (command === "enroll") {
21
+ process.stdout.write(
22
+ `${JSON.stringify({ capability: "voice-realtime", profile: await voice.enroll({ tenantId: "local" }) }, null, 2)}
23
+ `
24
+ );
25
+ } else if (command === "test-mic") {
26
+ process.stdout.write(
27
+ `${JSON.stringify({ capability: "voice-realtime", mic: await voice.testMic() }, null, 2)}
28
+ `
29
+ );
30
+ } else if (command === "narrate") {
31
+ const asset = await voice.synthesizeNarration(
32
+ { script: "Loop helps indie developers debug the work that matters.", targetDurationS: 3 },
33
+ createDemoBrandContext()
34
+ );
35
+ process.stdout.write(`${JSON.stringify({ capability: "voice-realtime", asset }, null, 2)}
36
+ `);
37
+ } else {
38
+ process.stderr.write(`Unknown voice-realtime command: ${command}
39
+ `);
40
+ process.exitCode = 1;
41
+ }
42
+ //# sourceMappingURL=cli.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/cli.ts"],"sourcesContent":["import { createDemoBrandContext } from \"@nebutra/generation-context\";\nimport { readVoiceDebug, VoiceRealtime } from \"./index\";\n\nconst command = process.argv[2] ?? \"doctor\";\nconst voice = new VoiceRealtime();\n\nif (command === \"doctor\") {\n process.stdout.write(\n `${JSON.stringify({ capability: \"voice-realtime\", results: await voice.doctor() }, null, 2)}\\n`,\n );\n} else if (command === \"debug\") {\n process.stdout.write(\n `${JSON.stringify({ capability: \"voice-realtime\", entries: await readVoiceDebug() }, null, 2)}\\n`,\n );\n} else if (command === \"enroll\") {\n process.stdout.write(\n `${JSON.stringify({ capability: \"voice-realtime\", profile: await voice.enroll({ tenantId: \"local\" }) }, null, 2)}\\n`,\n );\n} else if (command === \"test-mic\") {\n process.stdout.write(\n `${JSON.stringify({ capability: \"voice-realtime\", mic: await voice.testMic() }, null, 2)}\\n`,\n );\n} else if (command === \"narrate\") {\n const asset = await voice.synthesizeNarration(\n { script: \"Loop helps indie developers debug the work that matters.\", targetDurationS: 3 },\n createDemoBrandContext(),\n );\n process.stdout.write(`${JSON.stringify({ capability: \"voice-realtime\", asset }, null, 2)}\\n`);\n} else {\n process.stderr.write(`Unknown voice-realtime command: ${command}\\n`);\n process.exitCode = 1;\n}\n"],"mappings":";;;;;;AAAA,SAAS,8BAA8B;AAGvC,IAAM,UAAU,QAAQ,KAAK,CAAC,KAAK;AACnC,IAAM,QAAQ,IAAI,cAAc;AAEhC,IAAI,YAAY,UAAU;AACxB,UAAQ,OAAO;AAAA,IACb,GAAG,KAAK,UAAU,EAAE,YAAY,kBAAkB,SAAS,MAAM,MAAM,OAAO,EAAE,GAAG,MAAM,CAAC,CAAC;AAAA;AAAA,EAC7F;AACF,WAAW,YAAY,SAAS;AAC9B,UAAQ,OAAO;AAAA,IACb,GAAG,KAAK,UAAU,EAAE,YAAY,kBAAkB,SAAS,MAAM,eAAe,EAAE,GAAG,MAAM,CAAC,CAAC;AAAA;AAAA,EAC/F;AACF,WAAW,YAAY,UAAU;AAC/B,UAAQ,OAAO;AAAA,IACb,GAAG,KAAK,UAAU,EAAE,YAAY,kBAAkB,SAAS,MAAM,MAAM,OAAO,EAAE,UAAU,QAAQ,CAAC,EAAE,GAAG,MAAM,CAAC,CAAC;AAAA;AAAA,EAClH;AACF,WAAW,YAAY,YAAY;AACjC,UAAQ,OAAO;AAAA,IACb,GAAG,KAAK,UAAU,EAAE,YAAY,kBAAkB,KAAK,MAAM,MAAM,QAAQ,EAAE,GAAG,MAAM,CAAC,CAAC;AAAA;AAAA,EAC1F;AACF,WAAW,YAAY,WAAW;AAChC,QAAM,QAAQ,MAAM,MAAM;AAAA,IACxB,EAAE,QAAQ,4DAA4D,iBAAiB,EAAE;AAAA,IACzF,uBAAuB;AAAA,EACzB;AACA,UAAQ,OAAO,MAAM,GAAG,KAAK,UAAU,EAAE,YAAY,kBAAkB,MAAM,GAAG,MAAM,CAAC,CAAC;AAAA,CAAI;AAC9F,OAAO;AACL,UAAQ,OAAO,MAAM,mCAAmC,OAAO;AAAA,CAAI;AACnE,UAAQ,WAAW;AACrB;","names":[]}
@@ -0,0 +1,54 @@
1
+ import { GeneratedAsset, BrandContext } from '@nebutra/generation-context';
2
+
3
+ type VoiceState = "listening" | "thinking" | "speaking" | "paused" | "closed";
4
+ interface VoiceSession {
5
+ readonly id: string;
6
+ readonly tenantId: string;
7
+ readonly threadId: string;
8
+ readonly room: string;
9
+ readonly state: VoiceState;
10
+ }
11
+ interface NarrationRequest {
12
+ readonly script: string;
13
+ readonly voiceProfileId?: string;
14
+ readonly targetDurationS?: number;
15
+ }
16
+ interface VoiceAsset extends GeneratedAsset {
17
+ readonly kind: "voice";
18
+ readonly transcript: string;
19
+ readonly durationS: number;
20
+ readonly format: "wav";
21
+ readonly voiceProfileId: string;
22
+ }
23
+ interface VoiceProfile {
24
+ readonly id: string;
25
+ readonly tenantId: string;
26
+ readonly consentRecordedAt: string;
27
+ readonly sampleCount: number;
28
+ }
29
+ interface VoiceHealth {
30
+ readonly provider: string;
31
+ readonly ok: boolean;
32
+ readonly suggestion?: string;
33
+ }
34
+ interface VoiceRealtimeOptions {
35
+ readonly root?: string;
36
+ }
37
+ declare function readVoiceDebug(root?: string, limit?: number): Promise<unknown[]>;
38
+ declare class VoiceRealtime {
39
+ #private;
40
+ constructor(options?: VoiceRealtimeOptions);
41
+ startSession(input: {
42
+ tenantId: string;
43
+ threadId: string;
44
+ }): Promise<VoiceSession>;
45
+ synthesizeNarration(request: NarrationRequest, brandInput: BrandContext | undefined): Promise<VoiceAsset>;
46
+ enroll(input: {
47
+ tenantId: string;
48
+ samplePaths?: readonly string[];
49
+ }): Promise<VoiceProfile>;
50
+ testMic(): Promise<VoiceHealth>;
51
+ doctor(): Promise<VoiceHealth[]>;
52
+ }
53
+
54
+ export { type NarrationRequest, type VoiceAsset, type VoiceHealth, type VoiceProfile, VoiceRealtime, type VoiceRealtimeOptions, type VoiceSession, type VoiceState, readVoiceDebug };
package/dist/index.js ADDED
@@ -0,0 +1,9 @@
1
+ import {
2
+ VoiceRealtime,
3
+ readVoiceDebug
4
+ } from "./chunk-GARLTI7J.js";
5
+ export {
6
+ VoiceRealtime,
7
+ readVoiceDebug
8
+ };
9
+ //# sourceMappingURL=index.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
@@ -0,0 +1,5 @@
1
+ import { VoiceRealtime } from "../src/index";
2
+
3
+ const profile = await new VoiceRealtime().enroll({ tenantId: "demo", samplePaths: [] });
4
+
5
+ process.stdout.write(`${profile.id}\n`);
@@ -0,0 +1,9 @@
1
+ import { createDemoBrandContext } from "@nebutra/generation-context";
2
+ import { VoiceRealtime } from "../src/index";
3
+
4
+ const asset = await new VoiceRealtime().synthesizeNarration(
5
+ { script: "Loop makes debugging visible.", targetDurationS: 3 },
6
+ createDemoBrandContext(),
7
+ );
8
+
9
+ process.stdout.write(`${asset.path}\n`);
@@ -0,0 +1,5 @@
1
+ import { VoiceRealtime } from "../src/index";
2
+
3
+ process.stdout.write(
4
+ `${JSON.stringify(await new VoiceRealtime().startSession({ tenantId: "demo", threadId: "thread_1" }), null, 2)}\n`,
5
+ );
package/package.json ADDED
@@ -0,0 +1,61 @@
1
+ {
2
+ "name": "@nebutra/voice-realtime",
3
+ "version": "0.1.0",
4
+ "description": "Voice session lifecycle, narration synthesis, and enrollment surface",
5
+ "private": false,
6
+ "license": "MIT",
7
+ "type": "module",
8
+ "nebutra": {
9
+ "status": "wip",
10
+ "productionReady": false,
11
+ "surface": "generation-capability",
12
+ "requires": [
13
+ "@nebutra/generation-context for BrandContext",
14
+ "@nebutra/audio-pipeline for audio rendering utilities",
15
+ "A voice sidecar for production WebRTC/STT/TTS transport"
16
+ ],
17
+ "gaps": [
18
+ "Realtime transport is represented as a session contract until the sidecar is wired",
19
+ "Voice enrollment stores metadata only",
20
+ "Low-latency STT/TTS providers are health-checked through env only"
21
+ ],
22
+ "featureId": "voice-realtime",
23
+ "category": "ai",
24
+ "summary": "Realtime voice session and narration synthesis surface"
25
+ },
26
+ "main": "./src/index.ts",
27
+ "types": "./src/index.ts",
28
+ "exports": {
29
+ ".": "./src/index.ts"
30
+ },
31
+ "dependencies": {
32
+ "@nebutra/audio-pipeline": "0.1.0",
33
+ "@nebutra/capability-kit": "0.2.0",
34
+ "@nebutra/errors": "0.1.0",
35
+ "@nebutra/generation-context": "0.1.0"
36
+ },
37
+ "devDependencies": {
38
+ "@types/node": "^22.19.15",
39
+ "tsup": "^8.5.1",
40
+ "tsx": "^4.21.0",
41
+ "typescript": "^5.9.3",
42
+ "vitest": "^4.1.4"
43
+ },
44
+ "homepage": "https://github.com/Nebutra/Nebutra-Sailor/tree/main/packages/ai/voice-realtime#readme",
45
+ "repository": {
46
+ "type": "git",
47
+ "url": "git+https://github.com/Nebutra/Nebutra-Sailor.git",
48
+ "directory": "packages/ai/voice-realtime"
49
+ },
50
+ "bugs": {
51
+ "url": "https://github.com/Nebutra/Nebutra-Sailor/issues"
52
+ },
53
+ "publishConfig": {
54
+ "access": "public"
55
+ },
56
+ "scripts": {
57
+ "build": "tsup",
58
+ "test": "vitest run",
59
+ "typecheck": "tsc --noEmit"
60
+ }
61
+ }
package/src/cli.ts ADDED
@@ -0,0 +1,32 @@
1
+ import { createDemoBrandContext } from "@nebutra/generation-context";
2
+ import { readVoiceDebug, VoiceRealtime } from "./index";
3
+
4
+ const command = process.argv[2] ?? "doctor";
5
+ const voice = new VoiceRealtime();
6
+
7
+ if (command === "doctor") {
8
+ process.stdout.write(
9
+ `${JSON.stringify({ capability: "voice-realtime", results: await voice.doctor() }, null, 2)}\n`,
10
+ );
11
+ } else if (command === "debug") {
12
+ process.stdout.write(
13
+ `${JSON.stringify({ capability: "voice-realtime", entries: await readVoiceDebug() }, null, 2)}\n`,
14
+ );
15
+ } else if (command === "enroll") {
16
+ process.stdout.write(
17
+ `${JSON.stringify({ capability: "voice-realtime", profile: await voice.enroll({ tenantId: "local" }) }, null, 2)}\n`,
18
+ );
19
+ } else if (command === "test-mic") {
20
+ process.stdout.write(
21
+ `${JSON.stringify({ capability: "voice-realtime", mic: await voice.testMic() }, null, 2)}\n`,
22
+ );
23
+ } else if (command === "narrate") {
24
+ const asset = await voice.synthesizeNarration(
25
+ { script: "Loop helps indie developers debug the work that matters.", targetDurationS: 3 },
26
+ createDemoBrandContext(),
27
+ );
28
+ process.stdout.write(`${JSON.stringify({ capability: "voice-realtime", asset }, null, 2)}\n`);
29
+ } else {
30
+ process.stderr.write(`Unknown voice-realtime command: ${command}\n`);
31
+ process.exitCode = 1;
32
+ }
@@ -0,0 +1,33 @@
1
+ import { mkdtemp, readFile, rm } from "node:fs/promises";
2
+ import { tmpdir } from "node:os";
3
+ import { join } from "node:path";
4
+ import { createDemoBrandContext } from "@nebutra/generation-context";
5
+ import { afterEach, describe, expect, it } from "vitest";
6
+ import { readVoiceDebug, VoiceRealtime } from "./index";
7
+
8
+ let root: string | undefined;
9
+
10
+ afterEach(async () => {
11
+ if (root) await rm(root, { recursive: true, force: true });
12
+ root = undefined;
13
+ });
14
+
15
+ describe("VoiceRealtime", () => {
16
+ it("starts a thread-bound voice session without importing the runtime", async () => {
17
+ const session = await new VoiceRealtime().startSession({
18
+ tenantId: "tenant_a",
19
+ threadId: "thread_1",
20
+ });
21
+ expect(session.state).toBe("listening");
22
+ });
23
+
24
+ it("synthesizes narration with BrandContext", async () => {
25
+ root = await mkdtemp(join(tmpdir(), "voice-realtime-"));
26
+ const asset = await new VoiceRealtime({ root }).synthesizeNarration(
27
+ { script: "Ship the story with the founder voice.", targetDurationS: 2 },
28
+ createDemoBrandContext(),
29
+ );
30
+ expect((await readFile(asset.path)).subarray(0, 4).toString()).toBe("RIFF");
31
+ expect(await readVoiceDebug(root)).toHaveLength(1);
32
+ });
33
+ });
package/src/index.ts ADDED
@@ -0,0 +1,158 @@
1
+ import { mkdir, writeFile } from "node:fs/promises";
2
+ import { dirname, join } from "node:path";
3
+ import { createToneWav } from "@nebutra/audio-pipeline";
4
+ import { appendCapabilityDebug, readCapabilityDebug } from "@nebutra/capability-kit/debug";
5
+ import {
6
+ assetId,
7
+ type BrandContext,
8
+ type GeneratedAsset,
9
+ requireBrandContext,
10
+ } from "@nebutra/generation-context";
11
+
12
+ export type VoiceState = "listening" | "thinking" | "speaking" | "paused" | "closed";
13
+
14
+ export interface VoiceSession {
15
+ readonly id: string;
16
+ readonly tenantId: string;
17
+ readonly threadId: string;
18
+ readonly room: string;
19
+ readonly state: VoiceState;
20
+ }
21
+
22
+ export interface NarrationRequest {
23
+ readonly script: string;
24
+ readonly voiceProfileId?: string;
25
+ readonly targetDurationS?: number;
26
+ }
27
+
28
+ export interface VoiceAsset extends GeneratedAsset {
29
+ readonly kind: "voice";
30
+ readonly transcript: string;
31
+ readonly durationS: number;
32
+ readonly format: "wav";
33
+ readonly voiceProfileId: string;
34
+ }
35
+
36
+ export interface VoiceProfile {
37
+ readonly id: string;
38
+ readonly tenantId: string;
39
+ readonly consentRecordedAt: string;
40
+ readonly sampleCount: number;
41
+ }
42
+
43
+ export interface VoiceHealth {
44
+ readonly provider: string;
45
+ readonly ok: boolean;
46
+ readonly suggestion?: string;
47
+ }
48
+
49
+ export interface VoiceRealtimeOptions {
50
+ readonly root?: string;
51
+ }
52
+
53
+ export async function readVoiceDebug(root = process.cwd(), limit = 10): Promise<unknown[]> {
54
+ return readCapabilityDebug("voice-realtime", { root, limit });
55
+ }
56
+
57
+ export class VoiceRealtime {
58
+ readonly #root: string;
59
+
60
+ constructor(options: VoiceRealtimeOptions = {}) {
61
+ this.#root = options.root ?? process.cwd();
62
+ }
63
+
64
+ async startSession(input: { tenantId: string; threadId: string }): Promise<VoiceSession> {
65
+ const session: VoiceSession = {
66
+ id: assetId("voice_session", input.threadId),
67
+ tenantId: input.tenantId,
68
+ threadId: input.threadId,
69
+ room: `voice_${input.tenantId}_${input.threadId}`,
70
+ state: "listening",
71
+ };
72
+ await appendCapabilityDebug(
73
+ "voice-realtime",
74
+ { type: "session", session },
75
+ { root: this.#root },
76
+ );
77
+ return session;
78
+ }
79
+
80
+ async synthesizeNarration(
81
+ request: NarrationRequest,
82
+ brandInput: BrandContext | undefined,
83
+ ): Promise<VoiceAsset> {
84
+ const brand = requireBrandContext(brandInput, "voice-realtime");
85
+ const durationS = Math.max(
86
+ 1,
87
+ Math.min(request.targetDurationS ?? Math.ceil(request.script.length / 24), 20),
88
+ );
89
+ const id = assetId("voice", `${brand.brandId}_${request.script}`);
90
+ const path = join(this.#root, ".nebutra", "generated", "voice-realtime", `${id}.wav`);
91
+ await mkdir(dirname(path), { recursive: true });
92
+ await writeFile(path, createToneWav(durationS, 330));
93
+ const asset: VoiceAsset = {
94
+ id,
95
+ tenantId: brand.tenantId,
96
+ kind: "voice",
97
+ path,
98
+ brandId: brand.brandId,
99
+ provider: "tone-local",
100
+ model: "brand-context-narration-v1",
101
+ createdAt: new Date().toISOString(),
102
+ license: { status: "commercial-ok", source: "deterministic local renderer" },
103
+ transcript: request.script,
104
+ durationS,
105
+ format: "wav",
106
+ voiceProfileId: request.voiceProfileId ?? "default",
107
+ metadata: { brandSource: brand.sourcePath },
108
+ };
109
+ await appendCapabilityDebug(
110
+ "voice-realtime",
111
+ { type: "narration", asset },
112
+ { root: this.#root },
113
+ );
114
+ return asset;
115
+ }
116
+
117
+ async enroll(input: {
118
+ tenantId: string;
119
+ samplePaths?: readonly string[];
120
+ }): Promise<VoiceProfile> {
121
+ const profile: VoiceProfile = {
122
+ id: assetId("voice_profile", input.tenantId),
123
+ tenantId: input.tenantId,
124
+ consentRecordedAt: new Date().toISOString(),
125
+ sampleCount: input.samplePaths?.length ?? 0,
126
+ };
127
+ await appendCapabilityDebug(
128
+ "voice-realtime",
129
+ { type: "enroll", profile },
130
+ { root: this.#root },
131
+ );
132
+ return profile;
133
+ }
134
+
135
+ async testMic(): Promise<VoiceHealth> {
136
+ return {
137
+ provider: "local-mic",
138
+ ok: Boolean(process.env.VOICE_MIC_PERMISSION === "granted"),
139
+ suggestion: "Grant microphone access in the desktop app before realtime sessions.",
140
+ };
141
+ }
142
+
143
+ async doctor(): Promise<VoiceHealth[]> {
144
+ return [
145
+ await this.testMic(),
146
+ {
147
+ provider: "voice-sidecar",
148
+ ok: Boolean(process.env.VOICE_SIDECAR_URL),
149
+ suggestion: "Set VOICE_SIDECAR_URL to enable realtime WebRTC/STT/TTS.",
150
+ },
151
+ {
152
+ provider: "remote-tts",
153
+ ok: Boolean(process.env.VOICE_REMOTE_API_KEY),
154
+ suggestion: "Set VOICE_REMOTE_API_KEY to enable remote voice fallback.",
155
+ },
156
+ ];
157
+ }
158
+ }
package/tsconfig.json ADDED
@@ -0,0 +1,12 @@
1
+ {
2
+ "extends": "../../../tsconfig.base.json",
3
+ "compilerOptions": {
4
+ "module": "ESNext",
5
+ "moduleResolution": "bundler",
6
+ "target": "esnext",
7
+ "types": ["node"],
8
+ "incremental": false
9
+ },
10
+ "include": ["src", "examples"],
11
+ "exclude": ["node_modules", "dist"]
12
+ }
package/tsup.config.ts ADDED
@@ -0,0 +1,16 @@
1
+ import { defineConfig } from "tsup";
2
+
3
+ export default defineConfig({
4
+ entry: ["src/index.ts", "src/cli.ts"],
5
+ format: ["esm"],
6
+ dts: true,
7
+ sourcemap: true,
8
+ clean: true,
9
+ target: "es2022",
10
+ external: [
11
+ "@nebutra/audio-pipeline",
12
+ "@nebutra/capability-kit/debug",
13
+ "@nebutra/errors",
14
+ "@nebutra/generation-context",
15
+ ],
16
+ });