@frockbot/cloudflare 0.0.0 → 0.7.292
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build-artifact.ts +130 -0
- package/build-flutter-web.ts +508 -0
- package/deployment-config/README.md +414 -0
- package/deployment-config/cli.ts +228 -0
- package/deployment-config/generate.ts +689 -0
- package/deployment-config/jsonc.ts +18 -0
- package/deployment-config/profile-schema.generated.ts +391 -0
- package/deployment-config/profile.schema.json +358 -0
- package/deployment-config/profile.ts +124 -0
- package/migrations/0001_better_auth.sql +15 -0
- package/migrations/0002_drop_account_issuer.sql +8 -0
- package/package.json +61 -5
- package/release-version.ts +20 -0
- package/src/account-admission.ts +179 -0
- package/src/account-deletion.ts +274 -0
- package/src/admin-entrypoint.ts +194 -0
- package/src/admin-identities.ts +26 -0
- package/src/audit.ts +103 -0
- package/src/auth-package.access.ts +24 -0
- package/src/auth-package.ts +36 -0
- package/src/avatar-state-cleanup.ts +107 -0
- package/src/billing-computer.ts +267 -0
- package/src/billing-readiness.ts +21 -0
- package/src/billing.ts +461 -0
- package/src/bot-capabilities.ts +480 -0
- package/src/bot-recovery.ts +1 -0
- package/src/bot-state-channel.ts +702 -0
- package/src/bot-state.ts +3995 -0
- package/src/bot-template-cleanup.ts +98 -0
- package/src/bot-title-cleanup.ts +104 -0
- package/src/brand-icon.png +0 -0
- package/src/brand-logo.ts +2 -0
- package/src/brand.ts +36 -0
- package/src/client-compatibility.ts +51 -0
- package/src/compaction-announcement-cleanup.ts +113 -0
- package/src/computer-egress.ts +110 -0
- package/src/computer-host.ts +92 -0
- package/src/computer-screenshot-cleanup.ts +78 -0
- package/src/contracts.ts +832 -0
- package/src/debug.ts +272 -0
- package/src/default-packages-marker-cleanup.ts +39 -0
- package/src/deployment-policy-admin-host.ts +121 -0
- package/src/deployment-policy.ts +426 -0
- package/src/directory-profile-cleanup.ts +101 -0
- package/src/durable-rpc.ts +753 -0
- package/src/durable-session.ts +21 -0
- package/src/entry-boundary.ts +103 -0
- package/src/frock-ai.ts +338 -0
- package/src/gateway.ts +1392 -0
- package/src/group-chat.ts +736 -0
- package/src/hidden-bot-notifications-cleanup.ts +32 -0
- package/src/index.ts +2798 -0
- package/src/insights.ts +15 -0
- package/src/machine-messages-cleanup.ts +121 -0
- package/src/machine-socket.ts +164 -0
- package/src/memory-records.ts +181 -0
- package/src/memory.ts +145 -0
- package/src/model-rates.ts +252 -0
- package/src/native-auth.ts +1006 -0
- package/src/native-sessions.ts +121 -0
- package/src/notification-state-cleanup.ts +18 -0
- package/src/ollama-web-search-cleanup.ts +69 -0
- package/src/package-page-shapes-cleanup.ts +190 -0
- package/src/plugin-egress.ts +31 -0
- package/src/plugin-page-route.ts +72 -0
- package/src/plugin-panels-cleanup.ts +101 -0
- package/src/prepared-input-cleanup.ts +105 -0
- package/src/production-secrets.ts +484 -0
- package/src/project-cleanup.ts +123 -0
- package/src/project-events-cleanup.ts +152 -0
- package/src/publication-state-cleanup.ts +92 -0
- package/src/push.ts +498 -0
- package/src/request-body.ts +165 -0
- package/src/routine-state-cleanup.ts +57 -0
- package/src/search.ts +88 -0
- package/src/sidebar-label-cleanup.ts +165 -0
- package/src/skill-index-cleanup.ts +53 -0
- package/src/supersede-cleanup.ts +162 -0
- package/src/test-chat-cleanup.ts +93 -0
- package/src/uploads.ts +486 -0
- package/src/user-application.ts +1068 -0
- package/src/user-configuration.ts +4491 -0
- package/src/voice-assistant.ts +4999 -0
- package/src/voice-dictation.ts +743 -0
- package/src/working-context-cleanup.ts +96 -0
- package/src/workspace.ts +201 -0
- package/wrangler.jsonc +385 -0
- package/README.md +0 -3
|
@@ -0,0 +1,743 @@
|
|
|
1
|
+
// The composer's dictation relay: one client socket, one upstream socket.
|
|
2
|
+
//
|
|
3
|
+
// Deliberately not a Durable Object. Dictation is short-lived and admits
|
|
4
|
+
// nothing durable — the transcript lands in an editable draft and only the
|
|
5
|
+
// ordinary Send admits a Turn — and the one thing it must protect is the
|
|
6
|
+
// provider key, which stays here. The gateway proves the identity before
|
|
7
|
+
// this runs; this module never sees a cookie or a bearer token.
|
|
8
|
+
//
|
|
9
|
+
// What it does hold the line on:
|
|
10
|
+
//
|
|
11
|
+
// Spend. Before the upstream is opened the account's lease is taken from
|
|
12
|
+
// the voice object — one dictation at a time per account, a bounded window
|
|
13
|
+
// of seconds reserved up front and renewed while the capture runs — so a
|
|
14
|
+
// page that opens sockets in a loop is refused rather than billed.
|
|
15
|
+
//
|
|
16
|
+
// Completeness. The upstream has no turn detection — the streaming
|
|
17
|
+
// transcription models refuse it — so a capture is one item and the
|
|
18
|
+
// relay's commit after `stop` is the only thing that closes it. Deltas
|
|
19
|
+
// grow the draft while the person speaks; the committed item's transcript
|
|
20
|
+
// is the segment that replaces them. The relay still hands segments over
|
|
21
|
+
// in committed order and says `final` only once every committed item has
|
|
22
|
+
// answered, which costs nothing and holds if an upstream ever commits more
|
|
23
|
+
// than one. A stop the upstream cannot finish in time, or a provider
|
|
24
|
+
// failure after stop, is reported as what it is; the draft keeps what
|
|
25
|
+
// arrived.
|
|
26
|
+
//
|
|
27
|
+
// Opening audio. Frames that arrive before the upstream has accepted the
|
|
28
|
+
// session are held in order, bounded, and forwarded once it has, so
|
|
29
|
+
// pressing the microphone and speaking at once loses nothing.
|
|
30
|
+
//
|
|
31
|
+
// Tidying. Once every segment is in, and before `final`, the capture's own
|
|
32
|
+
// words are offered to Groq to have the fillers and false starts taken
|
|
33
|
+
// out, then to Jev to decide whether the tidy still says what they said.
|
|
34
|
+
// The client has already landed the raw segment by then. It happens
|
|
35
|
+
// here rather than on the client because both models are reached with a
|
|
36
|
+
// server-side credential, and it happens after the capture rather than
|
|
37
|
+
// during it because text that rewrites itself under the cursor is worse
|
|
38
|
+
// than text that is untidy. Every way this can go wrong — no model, no
|
|
39
|
+
// allowance, a timeout, a cheap refusal, a Jev `unfaithful` or a Jev
|
|
40
|
+
// miss — ends in the same place: `final`, with the raw transcript standing.
|
|
41
|
+
import type { DictationCleanupJudgeV1 } from "@frockbot/app/supervision";
|
|
42
|
+
import {
|
|
43
|
+
voiceDictationCleanupBodyV1,
|
|
44
|
+
voiceDictationCleanupResultV1,
|
|
45
|
+
voiceDictationCleanupWorthwhileV1,
|
|
46
|
+
} from "@frockbot/app/voice/dictation-cleanup";
|
|
47
|
+
import {
|
|
48
|
+
voiceDictationSessionUpdateV1,
|
|
49
|
+
voiceDictationUpstreamTargetV1,
|
|
50
|
+
type VoiceDictationEnvV1,
|
|
51
|
+
} from "@frockbot/app/voice/dictation-upstream";
|
|
52
|
+
import {
|
|
53
|
+
translateVoiceRealtimeUpstreamFrameV1,
|
|
54
|
+
voiceRealtimeAppendV1,
|
|
55
|
+
voiceRealtimeCommitV1,
|
|
56
|
+
} from "@frockbot/app/voice/openai-realtime";
|
|
57
|
+
import {
|
|
58
|
+
decodeVoiceDictationClientFrameV1,
|
|
59
|
+
VOICE_DICTATION_CLEANUP_TIMEOUT_MS_V1,
|
|
60
|
+
VOICE_DICTATION_FINAL_TIMEOUT_MS_V1,
|
|
61
|
+
VOICE_DICTATION_LEASE_RENEW_MS_V1,
|
|
62
|
+
VOICE_DICTATION_MAX_MS_V1,
|
|
63
|
+
VOICE_DICTATION_OPENING_BUFFER_BYTES_V1,
|
|
64
|
+
VOICE_REALTIME_CONNECT_TIMEOUT_MS_V1,
|
|
65
|
+
type VoiceDictationServerFrameV1,
|
|
66
|
+
} from "@frockbot/app/voice/shared";
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* The account's dictation lease, held by the voice object. Acquired before
|
|
70
|
+
* the provider is opened, renewed while the capture runs, released with the
|
|
71
|
+
* seconds actually used so the meter is reconciled.
|
|
72
|
+
*/
|
|
73
|
+
export interface VoiceDictationLeaseV1 {
|
|
74
|
+
acquire(): Promise<
|
|
75
|
+
{ status: "acquired" } | { status: "refused"; reason: string }
|
|
76
|
+
>;
|
|
77
|
+
/** False when the account has run out of allowance; the relay then stops. */
|
|
78
|
+
renew(): Promise<boolean>;
|
|
79
|
+
release(activeSeconds: number): Promise<void>;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Asks the model to tidy one transcript.
|
|
84
|
+
*
|
|
85
|
+
* Only the asking: booking the spend and reaching the gateway with a
|
|
86
|
+
* server-side credential. What we ask for and what we accept back are the
|
|
87
|
+
* relay's, so the policy is one testable place and the Worker's wiring has no
|
|
88
|
+
* judgement in it.
|
|
89
|
+
*/
|
|
90
|
+
export interface VoiceDictationCleanupV1 {
|
|
91
|
+
/**
|
|
92
|
+
* The model's answer, or nothing when the account has no allowance left.
|
|
93
|
+
* Throwing is the ordinary failure and is reported as "keep the raw text".
|
|
94
|
+
*/
|
|
95
|
+
run(
|
|
96
|
+
body: Record<string, unknown>,
|
|
97
|
+
signal: AbortSignal,
|
|
98
|
+
): Promise<string | undefined>;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export interface VoiceDictationRelayOptions {
|
|
102
|
+
env: VoiceDictationEnvV1;
|
|
103
|
+
lease?: VoiceDictationLeaseV1;
|
|
104
|
+
/** Absent in a deployment with no model gateway; the raw transcript stands. */
|
|
105
|
+
cleanup?: VoiceDictationCleanupV1;
|
|
106
|
+
/**
|
|
107
|
+
* Jev's rejector for a Groq tidy. Absent or unavailable keeps the raw
|
|
108
|
+
* transcript — a missing review is not an accept.
|
|
109
|
+
*/
|
|
110
|
+
cleanupJudge?: DictationCleanupJudgeV1;
|
|
111
|
+
cleanupTimeoutMs?: number;
|
|
112
|
+
/** Opens the upstream socket; the default is a `fetch` upgrade. */
|
|
113
|
+
connectUpstream?: (
|
|
114
|
+
url: string,
|
|
115
|
+
headers: Record<string, string>,
|
|
116
|
+
) => Promise<WebSocket>;
|
|
117
|
+
connectTimeoutMs?: number;
|
|
118
|
+
finalTimeoutMs?: number;
|
|
119
|
+
maxCaptureMs?: number;
|
|
120
|
+
leaseRenewMs?: number;
|
|
121
|
+
now?: () => number;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** Opens the upstream with a `fetch` upgrade, the way a Worker must. */
|
|
125
|
+
export async function fetchVoiceUpstreamSocketV1(
|
|
126
|
+
url: string,
|
|
127
|
+
headers: Record<string, string>,
|
|
128
|
+
signal?: AbortSignal,
|
|
129
|
+
): Promise<WebSocket> {
|
|
130
|
+
if (signal?.aborted) {
|
|
131
|
+
throw new DOMException("The operation was aborted.", "AbortError");
|
|
132
|
+
}
|
|
133
|
+
const target = new URL(url);
|
|
134
|
+
if (target.protocol === "wss:") target.protocol = "https:";
|
|
135
|
+
if (target.protocol === "ws:") target.protocol = "http:";
|
|
136
|
+
const response = await fetch(target, {
|
|
137
|
+
headers: { ...headers, upgrade: "websocket" },
|
|
138
|
+
...(signal ? { signal } : {}),
|
|
139
|
+
});
|
|
140
|
+
const socket = response.webSocket;
|
|
141
|
+
if (response.status !== 101 || !socket) {
|
|
142
|
+
throw new Error(`upstream refused the upgrade (${response.status})`);
|
|
143
|
+
}
|
|
144
|
+
if (signal?.aborted) {
|
|
145
|
+
try {
|
|
146
|
+
socket.close();
|
|
147
|
+
} catch {
|
|
148
|
+
// A late upgrade is closed rather than accepted.
|
|
149
|
+
}
|
|
150
|
+
throw new DOMException("The operation was aborted.", "AbortError");
|
|
151
|
+
}
|
|
152
|
+
socket.accept();
|
|
153
|
+
if (signal) {
|
|
154
|
+
const onAbort = () => {
|
|
155
|
+
try {
|
|
156
|
+
socket.close();
|
|
157
|
+
} catch {
|
|
158
|
+
// Already gone.
|
|
159
|
+
}
|
|
160
|
+
};
|
|
161
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
162
|
+
}
|
|
163
|
+
return socket;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Answers the authenticated `GET /api/voice/dictation` upgrade.
|
|
168
|
+
*
|
|
169
|
+
* The response is the 101 the gateway hands back; everything after it is the
|
|
170
|
+
* relay's own loop over the two sockets.
|
|
171
|
+
*/
|
|
172
|
+
export function openVoiceDictationRelayV1(
|
|
173
|
+
request: Request,
|
|
174
|
+
options: VoiceDictationRelayOptions,
|
|
175
|
+
): Response {
|
|
176
|
+
if (request.headers.get("upgrade")?.toLowerCase() !== "websocket") {
|
|
177
|
+
return Response.json(
|
|
178
|
+
{ error: "expected a WebSocket upgrade" },
|
|
179
|
+
{ status: 426 },
|
|
180
|
+
);
|
|
181
|
+
}
|
|
182
|
+
const pair = new WebSocketPair();
|
|
183
|
+
const [client, server] = Object.values(pair);
|
|
184
|
+
server.accept();
|
|
185
|
+
// Audio must arrive as bytes, not as a Blob to be read asynchronously,
|
|
186
|
+
// or frames could be forwarded out of order.
|
|
187
|
+
try {
|
|
188
|
+
(server as { binaryType?: string }).binaryType = "arraybuffer";
|
|
189
|
+
} catch {
|
|
190
|
+
// An older runtime without the setter still answers ArrayBuffers.
|
|
191
|
+
}
|
|
192
|
+
runRelay(server, options);
|
|
193
|
+
return new Response(null, { status: 101, webSocket: client });
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
/**
|
|
197
|
+
* A binary frame as bytes, whichever shape the runtime hands it over in.
|
|
198
|
+
* `instanceof ArrayBuffer` is not enough on its own: a buffer created in
|
|
199
|
+
* another realm (the test runner's module context, for one) fails it while
|
|
200
|
+
* still being exactly what it says it is.
|
|
201
|
+
*/
|
|
202
|
+
function binaryFrame(data: unknown): ArrayBuffer | undefined {
|
|
203
|
+
if (typeof data === "string") return undefined;
|
|
204
|
+
if (data instanceof ArrayBuffer) return data;
|
|
205
|
+
if (ArrayBuffer.isView(data)) {
|
|
206
|
+
return data.buffer.slice(
|
|
207
|
+
data.byteOffset,
|
|
208
|
+
data.byteOffset + data.byteLength,
|
|
209
|
+
) as ArrayBuffer;
|
|
210
|
+
}
|
|
211
|
+
if (Object.prototype.toString.call(data) === "[object ArrayBuffer]") {
|
|
212
|
+
return data as ArrayBuffer;
|
|
213
|
+
}
|
|
214
|
+
return undefined;
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
function send(socket: WebSocket, frame: VoiceDictationServerFrameV1): void {
|
|
218
|
+
try {
|
|
219
|
+
socket.send(JSON.stringify(frame));
|
|
220
|
+
} catch {
|
|
221
|
+
// Closed already; the close handler tears the rest down.
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
type FailureCode =
|
|
226
|
+
"unconfigured" | "upstream" | "timeout" | "limit" | "protocol";
|
|
227
|
+
|
|
228
|
+
function runRelay(
|
|
229
|
+
client: WebSocket,
|
|
230
|
+
options: VoiceDictationRelayOptions,
|
|
231
|
+
): void {
|
|
232
|
+
const connectTimeoutMs =
|
|
233
|
+
options.connectTimeoutMs ?? VOICE_REALTIME_CONNECT_TIMEOUT_MS_V1;
|
|
234
|
+
const finalTimeoutMs =
|
|
235
|
+
options.finalTimeoutMs ?? VOICE_DICTATION_FINAL_TIMEOUT_MS_V1;
|
|
236
|
+
const maxCaptureMs = options.maxCaptureMs ?? VOICE_DICTATION_MAX_MS_V1;
|
|
237
|
+
const leaseRenewMs =
|
|
238
|
+
options.leaseRenewMs ?? VOICE_DICTATION_LEASE_RENEW_MS_V1;
|
|
239
|
+
const cleanupTimeoutMs =
|
|
240
|
+
options.cleanupTimeoutMs ?? VOICE_DICTATION_CLEANUP_TIMEOUT_MS_V1;
|
|
241
|
+
const connectUpstream = options.connectUpstream ?? fetchVoiceUpstreamSocketV1;
|
|
242
|
+
const now = options.now ?? (() => Date.now());
|
|
243
|
+
|
|
244
|
+
let started = false;
|
|
245
|
+
let upstream: WebSocket | undefined;
|
|
246
|
+
let upstreamReady = false;
|
|
247
|
+
let upstreamOpenedAt: number | undefined;
|
|
248
|
+
/** The provider seconds this capture spent, frozen when its socket closes. */
|
|
249
|
+
let upstreamSeconds: number | undefined;
|
|
250
|
+
let closed = false;
|
|
251
|
+
let stopping = false;
|
|
252
|
+
/** The relay's own commit after `stop` has been sent, and then answered. */
|
|
253
|
+
let commitSent = false;
|
|
254
|
+
let connectTimer: ReturnType<typeof setTimeout> | undefined;
|
|
255
|
+
let stopCommitSettled = false;
|
|
256
|
+
/** The five-minute cap fired: the capture is finalised, then refused. */
|
|
257
|
+
let capped = false;
|
|
258
|
+
let leaseHeld = false;
|
|
259
|
+
let pending: ArrayBuffer[] = [];
|
|
260
|
+
let pendingBytes = 0;
|
|
261
|
+
let truncated = false;
|
|
262
|
+
const timers = new Set<ReturnType<typeof setTimeout>>();
|
|
263
|
+
|
|
264
|
+
// Items the upstream committed, in the order it committed them, and the
|
|
265
|
+
// transcriptions that have answered so far. Segments go to the client in
|
|
266
|
+
// committed order even when completions arrive out of it.
|
|
267
|
+
const committedOrder: string[] = [];
|
|
268
|
+
const answered = new Map<string, string>();
|
|
269
|
+
const delivered = new Set<string>();
|
|
270
|
+
/**
|
|
271
|
+
* Every segment handed to the client, in order: the capture's own words and
|
|
272
|
+
* exactly the span the client will replace if the tidy-up is accepted.
|
|
273
|
+
*/
|
|
274
|
+
const spoken: string[] = [];
|
|
275
|
+
/** The capture is ending; a second arrival must not race the first. */
|
|
276
|
+
let finishing = false;
|
|
277
|
+
/** The `stop` deadline, disarmed once the capture is safely ending. */
|
|
278
|
+
let finalTimer: ReturnType<typeof setTimeout> | undefined;
|
|
279
|
+
const outstanding = new Set<string>();
|
|
280
|
+
const partials = new Map<string, string>();
|
|
281
|
+
|
|
282
|
+
const after = (ms: number, run: () => void) => {
|
|
283
|
+
const timer = setTimeout(() => {
|
|
284
|
+
timers.delete(timer);
|
|
285
|
+
run();
|
|
286
|
+
}, ms);
|
|
287
|
+
timers.add(timer);
|
|
288
|
+
return timer;
|
|
289
|
+
};
|
|
290
|
+
|
|
291
|
+
const activeSeconds = () =>
|
|
292
|
+
upstreamSeconds ??
|
|
293
|
+
(upstreamOpenedAt === undefined ? 0 : (now() - upstreamOpenedAt) / 1000);
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* Stops the provider clock and lets go of its socket. Called the moment the
|
|
297
|
+
* capture stops producing words, so the seconds the account is charged are
|
|
298
|
+
* the seconds a microphone was open — not the few a tidy-up spends after it.
|
|
299
|
+
*/
|
|
300
|
+
const closeUpstream = () => {
|
|
301
|
+
if (upstreamSeconds === undefined) upstreamSeconds = activeSeconds();
|
|
302
|
+
try {
|
|
303
|
+
upstream?.close(1000, "dictation ended");
|
|
304
|
+
} catch {
|
|
305
|
+
// Already closed.
|
|
306
|
+
}
|
|
307
|
+
};
|
|
308
|
+
|
|
309
|
+
const finish = (frame?: VoiceDictationServerFrameV1) => {
|
|
310
|
+
if (closed) return;
|
|
311
|
+
closed = true;
|
|
312
|
+
if (frame) send(client, frame);
|
|
313
|
+
for (const timer of timers) clearTimeout(timer);
|
|
314
|
+
timers.clear();
|
|
315
|
+
closeUpstream();
|
|
316
|
+
try {
|
|
317
|
+
client.close(1000, "dictation ended");
|
|
318
|
+
} catch {
|
|
319
|
+
// Already closed.
|
|
320
|
+
}
|
|
321
|
+
if (leaseHeld) {
|
|
322
|
+
leaseHeld = false;
|
|
323
|
+
void options.lease?.release(activeSeconds()).catch(() => undefined);
|
|
324
|
+
}
|
|
325
|
+
};
|
|
326
|
+
|
|
327
|
+
const fail = (message: string, code: FailureCode) =>
|
|
328
|
+
finish({ schemaVersion: 1, type: "error", message, code });
|
|
329
|
+
|
|
330
|
+
const forward = (chunk: ArrayBuffer) => {
|
|
331
|
+
if (upstream && upstreamReady) {
|
|
332
|
+
try {
|
|
333
|
+
upstream.send(voiceRealtimeAppendV1(chunk));
|
|
334
|
+
} catch {
|
|
335
|
+
fail("Dictation stopped: the speech service went away.", "upstream");
|
|
336
|
+
}
|
|
337
|
+
return;
|
|
338
|
+
}
|
|
339
|
+
pending.push(chunk);
|
|
340
|
+
pendingBytes += chunk.byteLength;
|
|
341
|
+
while (
|
|
342
|
+
pendingBytes > VOICE_DICTATION_OPENING_BUFFER_BYTES_V1 &&
|
|
343
|
+
pending.length > 1
|
|
344
|
+
) {
|
|
345
|
+
pendingBytes -= pending.shift()!.byteLength;
|
|
346
|
+
truncated = true;
|
|
347
|
+
}
|
|
348
|
+
};
|
|
349
|
+
|
|
350
|
+
/** Hands every answered item at the head of the committed order to the client. */
|
|
351
|
+
const flushSegments = () => {
|
|
352
|
+
while (committedOrder.length > 0) {
|
|
353
|
+
const head = committedOrder[0]!;
|
|
354
|
+
const text = answered.get(head);
|
|
355
|
+
if (text === undefined) break;
|
|
356
|
+
committedOrder.shift();
|
|
357
|
+
answered.delete(head);
|
|
358
|
+
delivered.add(head);
|
|
359
|
+
partials.delete(head);
|
|
360
|
+
if (text) {
|
|
361
|
+
spoken.push(text);
|
|
362
|
+
send(client, { schemaVersion: 1, type: "segment", text });
|
|
363
|
+
}
|
|
364
|
+
}
|
|
365
|
+
if (partials.size > 0)
|
|
366
|
+
send(client, {
|
|
367
|
+
schemaVersion: 1,
|
|
368
|
+
type: "delta",
|
|
369
|
+
text: [...partials.values()].join(" "),
|
|
370
|
+
});
|
|
371
|
+
};
|
|
372
|
+
|
|
373
|
+
const stopIsComplete = () =>
|
|
374
|
+
stopping && stopCommitSettled && outstanding.size === 0;
|
|
375
|
+
|
|
376
|
+
/**
|
|
377
|
+
* Ends a capture the upstream finished. A capture the five-minute cap
|
|
378
|
+
* stopped keeps every segment it produced and closes on the `limit` error
|
|
379
|
+
* in place of `final`, so the person is told why dictation ended — and it
|
|
380
|
+
* is not tidied, because a capture that was cut off mid-sentence is not one
|
|
381
|
+
* whose false starts we can tell from its words.
|
|
382
|
+
*/
|
|
383
|
+
const finishCapture = () => {
|
|
384
|
+
if (finishing || closed) return;
|
|
385
|
+
finishing = true;
|
|
386
|
+
// The capture is ending on its own terms now, so the stop deadline has
|
|
387
|
+
// nothing left to protect — and left armed it would report a timeout over
|
|
388
|
+
// the top of a tidy-up that is merely taking its few seconds.
|
|
389
|
+
if (finalTimer !== undefined) {
|
|
390
|
+
clearTimeout(finalTimer);
|
|
391
|
+
timers.delete(finalTimer);
|
|
392
|
+
finalTimer = undefined;
|
|
393
|
+
}
|
|
394
|
+
closeUpstream();
|
|
395
|
+
if (capped) {
|
|
396
|
+
fail(
|
|
397
|
+
"Dictation stopped after five minutes. Press the microphone to continue.",
|
|
398
|
+
"limit",
|
|
399
|
+
);
|
|
400
|
+
return;
|
|
401
|
+
}
|
|
402
|
+
void tidyThenFinish();
|
|
403
|
+
};
|
|
404
|
+
|
|
405
|
+
/**
|
|
406
|
+
* Offers the capture's words to the model, then ends the capture.
|
|
407
|
+
*
|
|
408
|
+
* Every path through this ends on `final`. A deployment with no gateway, an
|
|
409
|
+
* account out of allowance, a transcript too short or too long to be worth
|
|
410
|
+
* a call, a model that fails or takes too long, a cheap refusal, Jev
|
|
411
|
+
* refusing or Jev missing — all of them leave the raw transcript exactly
|
|
412
|
+
* where it already is, which is in the person's draft. Nothing here can
|
|
413
|
+
* lose text.
|
|
414
|
+
*/
|
|
415
|
+
const tidyThenFinish = async () => {
|
|
416
|
+
const done = () => finish({ schemaVersion: 1, type: "final" });
|
|
417
|
+
const cleanup = options.cleanup;
|
|
418
|
+
const transcript = spoken.join(" ").trim();
|
|
419
|
+
if (!cleanup || !voiceDictationCleanupWorthwhileV1(transcript)) {
|
|
420
|
+
done();
|
|
421
|
+
return;
|
|
422
|
+
}
|
|
423
|
+
send(client, { schemaVersion: 1, type: "cleaning" });
|
|
424
|
+
const deadline = new AbortController();
|
|
425
|
+
const timer = after(cleanupTimeoutMs, () =>
|
|
426
|
+
deadline.abort(new Error("the tidy-up took too long")),
|
|
427
|
+
);
|
|
428
|
+
try {
|
|
429
|
+
const answer = await cleanup.run(
|
|
430
|
+
voiceDictationCleanupBodyV1(transcript),
|
|
431
|
+
deadline.signal,
|
|
432
|
+
);
|
|
433
|
+
// The person closed the composer, sent, or navigated away while we
|
|
434
|
+
// asked. Their draft is not ours to touch any more.
|
|
435
|
+
if (closed) return;
|
|
436
|
+
if (answer !== undefined) {
|
|
437
|
+
const result = voiceDictationCleanupResultV1(transcript, answer);
|
|
438
|
+
if (result.status === "kept") {
|
|
439
|
+
// Named rather than silent: "cleanup is off" and "cleanup keeps
|
|
440
|
+
// eating people's negations" look identical without this line.
|
|
441
|
+
console.log("voice dictation cleanup kept the raw transcript", {
|
|
442
|
+
reason: result.reason,
|
|
443
|
+
});
|
|
444
|
+
} else {
|
|
445
|
+
const judge = options.cleanupJudge;
|
|
446
|
+
const verdict = judge
|
|
447
|
+
? await judge.review(
|
|
448
|
+
{ raw: transcript, tidied: result.text },
|
|
449
|
+
deadline.signal,
|
|
450
|
+
)
|
|
451
|
+
: "unavailable";
|
|
452
|
+
if (closed) return;
|
|
453
|
+
if (verdict === "faithful") {
|
|
454
|
+
send(client, {
|
|
455
|
+
schemaVersion: 1,
|
|
456
|
+
type: "cleaned",
|
|
457
|
+
text: result.text,
|
|
458
|
+
});
|
|
459
|
+
} else {
|
|
460
|
+
console.log("voice dictation cleanup kept the raw transcript", {
|
|
461
|
+
reason: verdict,
|
|
462
|
+
});
|
|
463
|
+
}
|
|
464
|
+
}
|
|
465
|
+
}
|
|
466
|
+
} catch (error) {
|
|
467
|
+
console.error("voice dictation cleanup failed", error);
|
|
468
|
+
} finally {
|
|
469
|
+
clearTimeout(timer);
|
|
470
|
+
timers.delete(timer);
|
|
471
|
+
}
|
|
472
|
+
if (closed) return;
|
|
473
|
+
done();
|
|
474
|
+
};
|
|
475
|
+
|
|
476
|
+
const finishIfComplete = () => {
|
|
477
|
+
if (stopIsComplete()) finishCapture();
|
|
478
|
+
};
|
|
479
|
+
|
|
480
|
+
const onUpstreamEvent = (raw: string) => {
|
|
481
|
+
const event = translateVoiceRealtimeUpstreamFrameV1(raw);
|
|
482
|
+
if (!event) return;
|
|
483
|
+
switch (event.kind) {
|
|
484
|
+
case "session-updated":
|
|
485
|
+
if (!upstreamReady) acceptSession();
|
|
486
|
+
return;
|
|
487
|
+
case "delta": {
|
|
488
|
+
const id = event.itemId ?? "uncommitted";
|
|
489
|
+
if (delivered.has(id)) return;
|
|
490
|
+
partials.set(id, (partials.get(id) ?? "") + event.text);
|
|
491
|
+
send(client, {
|
|
492
|
+
schemaVersion: 1,
|
|
493
|
+
type: "delta",
|
|
494
|
+
text: [...partials.values()].join(" "),
|
|
495
|
+
});
|
|
496
|
+
return;
|
|
497
|
+
}
|
|
498
|
+
case "committed":
|
|
499
|
+
if (!delivered.has(event.itemId)) {
|
|
500
|
+
if (!committedOrder.includes(event.itemId)) {
|
|
501
|
+
committedOrder.push(event.itemId);
|
|
502
|
+
}
|
|
503
|
+
if (!answered.has(event.itemId)) outstanding.add(event.itemId);
|
|
504
|
+
}
|
|
505
|
+
// Turn detection is off upstream, so the only thing that commits an
|
|
506
|
+
// item is the relay's own commit after `stop`.
|
|
507
|
+
if (stopping && commitSent) stopCommitSettled = true;
|
|
508
|
+
flushSegments();
|
|
509
|
+
return;
|
|
510
|
+
case "completed": {
|
|
511
|
+
const id = event.itemId ?? `unnamed-${answered.size + delivered.size}`;
|
|
512
|
+
outstanding.delete(id);
|
|
513
|
+
if (!committedOrder.includes(id) && !delivered.has(id)) {
|
|
514
|
+
committedOrder.push(id);
|
|
515
|
+
}
|
|
516
|
+
answered.set(id, event.text);
|
|
517
|
+
flushSegments();
|
|
518
|
+
finishIfComplete();
|
|
519
|
+
return;
|
|
520
|
+
}
|
|
521
|
+
case "failed": {
|
|
522
|
+
// One item the provider could not transcribe. Its words are lost and
|
|
523
|
+
// the person is told; the rest of the capture stands.
|
|
524
|
+
const id = event.itemId;
|
|
525
|
+
if (id) {
|
|
526
|
+
outstanding.delete(id);
|
|
527
|
+
if (!delivered.has(id)) {
|
|
528
|
+
if (!committedOrder.includes(id)) committedOrder.push(id);
|
|
529
|
+
answered.set(id, "");
|
|
530
|
+
}
|
|
531
|
+
}
|
|
532
|
+
console.error("voice dictation item failed", event.message);
|
|
533
|
+
send(client, {
|
|
534
|
+
schemaVersion: 1,
|
|
535
|
+
type: "notice",
|
|
536
|
+
message: "Part of what you said could not be transcribed.",
|
|
537
|
+
});
|
|
538
|
+
flushSegments();
|
|
539
|
+
finishIfComplete();
|
|
540
|
+
return;
|
|
541
|
+
}
|
|
542
|
+
case "error":
|
|
543
|
+
if (stopping && event.emptyBuffer) {
|
|
544
|
+
// The relay committed a buffer with nothing in it: the person
|
|
545
|
+
// pressed stop without saying anything new.
|
|
546
|
+
stopCommitSettled = true;
|
|
547
|
+
finishIfComplete();
|
|
548
|
+
return;
|
|
549
|
+
}
|
|
550
|
+
console.error("voice dictation upstream error", event.message);
|
|
551
|
+
fail(
|
|
552
|
+
stopping
|
|
553
|
+
? "Dictation ended before the last words were transcribed. What arrived is in your draft."
|
|
554
|
+
: "Dictation stopped: the speech service refused the session. Try again.",
|
|
555
|
+
"upstream",
|
|
556
|
+
);
|
|
557
|
+
return;
|
|
558
|
+
}
|
|
559
|
+
};
|
|
560
|
+
|
|
561
|
+
const start = async () => {
|
|
562
|
+
const target = voiceDictationUpstreamTargetV1(options.env);
|
|
563
|
+
if (target.path === "unconfigured") {
|
|
564
|
+
fail(target.message, "unconfigured");
|
|
565
|
+
return;
|
|
566
|
+
}
|
|
567
|
+
if (options.lease) {
|
|
568
|
+
let admission: Awaited<ReturnType<VoiceDictationLeaseV1["acquire"]>>;
|
|
569
|
+
try {
|
|
570
|
+
admission = await options.lease.acquire();
|
|
571
|
+
} catch (error) {
|
|
572
|
+
console.error("voice dictation lease unavailable", error);
|
|
573
|
+
fail("Dictation is unavailable right now. Try again.", "limit");
|
|
574
|
+
return;
|
|
575
|
+
}
|
|
576
|
+
if (closed) return;
|
|
577
|
+
if (admission.status === "refused") {
|
|
578
|
+
fail(admission.reason, "limit");
|
|
579
|
+
return;
|
|
580
|
+
}
|
|
581
|
+
leaseHeld = true;
|
|
582
|
+
const renew = () =>
|
|
583
|
+
after(leaseRenewMs, async () => {
|
|
584
|
+
if (closed) return;
|
|
585
|
+
let ok = false;
|
|
586
|
+
try {
|
|
587
|
+
ok = await options.lease!.renew();
|
|
588
|
+
} catch {
|
|
589
|
+
ok = false;
|
|
590
|
+
}
|
|
591
|
+
if (closed) return;
|
|
592
|
+
if (!ok) {
|
|
593
|
+
fail(
|
|
594
|
+
"Today's dictation allowance is used up. What arrived is in your draft.",
|
|
595
|
+
"limit",
|
|
596
|
+
);
|
|
597
|
+
return;
|
|
598
|
+
}
|
|
599
|
+
renew();
|
|
600
|
+
});
|
|
601
|
+
renew();
|
|
602
|
+
}
|
|
603
|
+
connectTimer = after(connectTimeoutMs, () =>
|
|
604
|
+
fail("Dictation didn't start in time. Try again.", "timeout"),
|
|
605
|
+
);
|
|
606
|
+
let socket: WebSocket;
|
|
607
|
+
try {
|
|
608
|
+
socket = await connectUpstream(target.url, target.headers);
|
|
609
|
+
} catch (error) {
|
|
610
|
+
console.error("voice dictation upstream refused", error);
|
|
611
|
+
fail(
|
|
612
|
+
"Dictation stopped: the speech service refused the connection. Try again.",
|
|
613
|
+
"upstream",
|
|
614
|
+
);
|
|
615
|
+
return;
|
|
616
|
+
}
|
|
617
|
+
if (closed) {
|
|
618
|
+
socket.close(1000, "client left");
|
|
619
|
+
return;
|
|
620
|
+
}
|
|
621
|
+
upstream = socket;
|
|
622
|
+
upstreamOpenedAt = now();
|
|
623
|
+
socket.addEventListener("message", (event) => {
|
|
624
|
+
if (typeof event.data !== "string") return;
|
|
625
|
+
onUpstreamEvent(event.data);
|
|
626
|
+
});
|
|
627
|
+
socket.addEventListener("close", () => {
|
|
628
|
+
if (closed) return;
|
|
629
|
+
if (stopIsComplete()) {
|
|
630
|
+
finishCapture();
|
|
631
|
+
return;
|
|
632
|
+
}
|
|
633
|
+
fail(
|
|
634
|
+
stopping
|
|
635
|
+
? "Dictation ended before the last words were transcribed. What arrived is in your draft."
|
|
636
|
+
: "Dictation stopped: the speech service closed the session.",
|
|
637
|
+
"upstream",
|
|
638
|
+
);
|
|
639
|
+
});
|
|
640
|
+
socket.addEventListener("error", () => {
|
|
641
|
+
if (!closed)
|
|
642
|
+
fail("Dictation stopped: the speech service failed.", "upstream");
|
|
643
|
+
});
|
|
644
|
+
socket.send(JSON.stringify(voiceDictationSessionUpdateV1()));
|
|
645
|
+
};
|
|
646
|
+
|
|
647
|
+
const acceptSession = () => {
|
|
648
|
+
if (connectTimer !== undefined) {
|
|
649
|
+
clearTimeout(connectTimer);
|
|
650
|
+
timers.delete(connectTimer);
|
|
651
|
+
}
|
|
652
|
+
upstreamReady = true;
|
|
653
|
+
const held = pending;
|
|
654
|
+
pending = [];
|
|
655
|
+
pendingBytes = 0;
|
|
656
|
+
for (const chunk of held) forward(chunk);
|
|
657
|
+
if (truncated) {
|
|
658
|
+
send(client, {
|
|
659
|
+
schemaVersion: 1,
|
|
660
|
+
type: "notice",
|
|
661
|
+
message:
|
|
662
|
+
"The first part of what you said was lost while dictation was starting.",
|
|
663
|
+
});
|
|
664
|
+
}
|
|
665
|
+
send(client, { schemaVersion: 1, type: "ready" });
|
|
666
|
+
after(maxCaptureMs, () => {
|
|
667
|
+
if (closed || stopping) return;
|
|
668
|
+
capped = true;
|
|
669
|
+
stop();
|
|
670
|
+
});
|
|
671
|
+
if (stopping) commit();
|
|
672
|
+
};
|
|
673
|
+
|
|
674
|
+
const stop = () => {
|
|
675
|
+
if (stopping) return;
|
|
676
|
+
stopping = true;
|
|
677
|
+
// Bounded from the moment of the stop, whether or not the upstream has
|
|
678
|
+
// opened yet: audio held here is still sent once it opens and committed
|
|
679
|
+
// then, and a stop the upstream cannot finish is reported, not faked.
|
|
680
|
+
finalTimer = after(finalTimeoutMs, () => {
|
|
681
|
+
if (closed || finishing) return;
|
|
682
|
+
fail(
|
|
683
|
+
"Dictation ended before the last words were transcribed. What arrived is in your draft.",
|
|
684
|
+
"timeout",
|
|
685
|
+
);
|
|
686
|
+
});
|
|
687
|
+
if (upstream && upstreamReady) commit();
|
|
688
|
+
};
|
|
689
|
+
|
|
690
|
+
const commit = () => {
|
|
691
|
+
if (commitSent) return;
|
|
692
|
+
commitSent = true;
|
|
693
|
+
try {
|
|
694
|
+
upstream!.send(voiceRealtimeCommitV1());
|
|
695
|
+
} catch {
|
|
696
|
+
fail(
|
|
697
|
+
"Dictation ended before the last words were transcribed. What arrived is in your draft.",
|
|
698
|
+
"upstream",
|
|
699
|
+
);
|
|
700
|
+
}
|
|
701
|
+
};
|
|
702
|
+
|
|
703
|
+
client.addEventListener("message", (event) => {
|
|
704
|
+
if (closed) return;
|
|
705
|
+
const binary = binaryFrame(event.data);
|
|
706
|
+
if (binary) {
|
|
707
|
+
if (!started) {
|
|
708
|
+
fail("Send the start frame before audio.", "protocol");
|
|
709
|
+
return;
|
|
710
|
+
}
|
|
711
|
+
if (stopping) return;
|
|
712
|
+
forward(binary);
|
|
713
|
+
return;
|
|
714
|
+
}
|
|
715
|
+
if (typeof event.data !== "string") return;
|
|
716
|
+
let frame;
|
|
717
|
+
try {
|
|
718
|
+
frame = decodeVoiceDictationClientFrameV1(JSON.parse(event.data));
|
|
719
|
+
} catch {
|
|
720
|
+
fail("That dictation frame is not understood.", "protocol");
|
|
721
|
+
return;
|
|
722
|
+
}
|
|
723
|
+
if (frame.type === "start") {
|
|
724
|
+
if (started) return;
|
|
725
|
+
started = true;
|
|
726
|
+
void start();
|
|
727
|
+
return;
|
|
728
|
+
}
|
|
729
|
+
if (frame.type === "stop") {
|
|
730
|
+
if (!started) {
|
|
731
|
+
finish({ schemaVersion: 1, type: "final" });
|
|
732
|
+
return;
|
|
733
|
+
}
|
|
734
|
+
stop();
|
|
735
|
+
}
|
|
736
|
+
});
|
|
737
|
+
client.addEventListener("close", () => {
|
|
738
|
+
finish();
|
|
739
|
+
});
|
|
740
|
+
client.addEventListener("error", () => {
|
|
741
|
+
finish();
|
|
742
|
+
});
|
|
743
|
+
}
|