@mentra/cloud-protocol 3.2.1-beta.631 → 3.2.1-beta.641
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/audio.ts +11 -0
- package/src/handshake.ts +5 -1
package/package.json
CHANGED
package/src/audio.ts
CHANGED
|
@@ -56,6 +56,13 @@ export type UdpLivenessAckPayload = z.infer<typeof udpLivenessAckPayloadSchema>;
|
|
|
56
56
|
|
|
57
57
|
// --- Result types -----------------------------------------------------------
|
|
58
58
|
|
|
59
|
+
/** Position on the phone's submitted audio stream, independent of provider latency. */
|
|
60
|
+
export const audioPositionSchema = z.object({
|
|
61
|
+
sessionTag: z.number().int().nonnegative(),
|
|
62
|
+
offsetMs: z.number().nonnegative(),
|
|
63
|
+
});
|
|
64
|
+
export type AudioPosition = z.infer<typeof audioPositionSchema>;
|
|
65
|
+
|
|
59
66
|
export const transcriptionTokenSchema = z.object({
|
|
60
67
|
text: z.string(),
|
|
61
68
|
startMs: z.number(),
|
|
@@ -65,6 +72,7 @@ export const transcriptionTokenSchema = z.object({
|
|
|
65
72
|
speaker: z.string().optional(),
|
|
66
73
|
// per-token; mid-sentence language switches are possible
|
|
67
74
|
detectedLanguage: z.string().optional(),
|
|
75
|
+
audioPosition: audioPositionSchema.optional(),
|
|
68
76
|
});
|
|
69
77
|
export type TranscriptionToken = z.infer<typeof transcriptionTokenSchema>;
|
|
70
78
|
|
|
@@ -88,6 +96,9 @@ export const transcriptionDataSchema = z.object({
|
|
|
88
96
|
|
|
89
97
|
// Per-token detail for consumers that need it
|
|
90
98
|
tokens: z.array(transcriptionTokenSchema),
|
|
99
|
+
// Internal result capability: missing timing on a positioned session must
|
|
100
|
+
// be withheld, while an older/unnegotiated session keeps legacy delivery.
|
|
101
|
+
frameTimelineVersion: z.literal(1).optional(),
|
|
91
102
|
|
|
92
103
|
provider: z.string(),
|
|
93
104
|
timestamp: z.number(),
|
package/src/handshake.ts
CHANGED
|
@@ -26,9 +26,12 @@ export const connectionInitPayloadSchema = z.object({
|
|
|
26
26
|
.object({
|
|
27
27
|
codec: z.enum(["lc3", "pcm"]),
|
|
28
28
|
sampleRate: z.number().int(), // e.g. 16000
|
|
29
|
+
frameTimelineVersion: z.literal(1).optional(),
|
|
29
30
|
// LC3 bitrate / frame size in bytes. Only meaningful when codec is
|
|
30
31
|
// "lc3"; omitted PCM sessions keep treating payload bytes as PCM.
|
|
31
|
-
frameSizeBytes: z
|
|
32
|
+
frameSizeBytes: z
|
|
33
|
+
.union([z.literal(20), z.literal(40), z.literal(60)])
|
|
34
|
+
.optional(),
|
|
32
35
|
// Optional initial subscription set, seeded atomically with the session
|
|
33
36
|
// so audio that starts before the first REST update is not transcribed
|
|
34
37
|
// with an empty set.
|
|
@@ -47,6 +50,7 @@ export const connectionAckPayloadSchema = z.object({
|
|
|
47
50
|
audio: z
|
|
48
51
|
.object({
|
|
49
52
|
sessionTag: z.number().int(), // u32 stamped into UDP audio frames
|
|
53
|
+
frameTimelineVersion: z.literal(1).optional(),
|
|
50
54
|
udp: z.object({ host: z.string(), port: z.number().int() }),
|
|
51
55
|
// Per-session key for encrypting UDP audio. Delivered here because the
|
|
52
56
|
// handshake is over the TLS WebSocket.
|