@avocadostudio-ai/orchestrator-core 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/sites-agent-context.js +3 -2
- package/dist/agent/sites-agent-shared.js +1 -0
- package/dist/chat/anthropic-planner.js +3 -3
- package/dist/chat/chat-pipeline.js +122 -20
- package/dist/chat/gemini-planner.js +3 -3
- package/dist/chat/planner.js +7 -5
- package/dist/chat/prompts.js +6 -1
- package/dist/cms/adapter.d.ts +159 -1
- package/dist/cms/adapter.js +19 -1
- package/dist/cms/bootstrap.d.ts +46 -1
- package/dist/cms/bootstrap.js +126 -2
- package/dist/cms/index.d.ts +3 -2
- package/dist/cms/index.js +2 -1
- package/dist/errors.d.ts +9 -1
- package/dist/handler/auth.d.ts +79 -0
- package/dist/handler/auth.js +113 -0
- package/dist/handler/create-orchestrator.d.ts +205 -0
- package/dist/handler/create-orchestrator.js +1599 -0
- package/dist/http/access-tokens.d.ts +58 -0
- package/dist/http/access-tokens.js +161 -0
- package/dist/http/audio-actions.d.ts +121 -0
- package/dist/http/audio-actions.js +248 -0
- package/dist/http/blocks-actions.d.ts +31 -0
- package/dist/http/blocks-actions.js +31 -0
- package/dist/http/draft-provenance.d.ts +68 -0
- package/dist/http/draft-provenance.js +101 -0
- package/dist/http/history-actions.d.ts +58 -0
- package/dist/http/history-actions.js +169 -0
- package/dist/http/image-generate-actions.d.ts +268 -0
- package/dist/http/image-generate-actions.js +546 -0
- package/dist/http/ops-actions.d.ts +51 -0
- package/dist/http/ops-actions.js +79 -0
- package/dist/http/publish-actions.d.ts +153 -0
- package/dist/http/publish-actions.js +323 -0
- package/dist/http/restore-actions.d.ts +67 -0
- package/dist/http/restore-actions.js +145 -0
- package/dist/http/screenshot-actions.d.ts +108 -0
- package/dist/http/screenshot-actions.js +181 -0
- package/dist/http/session-actions.d.ts +35 -0
- package/dist/http/session-actions.js +98 -0
- package/dist/http/telemetry-feedback-actions.d.ts +53 -0
- package/dist/http/telemetry-feedback-actions.js +68 -0
- package/dist/http/unsplash-actions.d.ts +64 -0
- package/dist/http/unsplash-actions.js +81 -0
- package/dist/http/variations-actions.d.ts +102 -0
- package/dist/http/variations-actions.js +104 -0
- package/dist/index.d.ts +4 -1
- package/dist/index.js +21 -1
- package/dist/nlp/deterministic-planner-refs.d.ts +1 -1
- package/dist/nlp/deterministic-planner-suggestions.d.ts +10 -0
- package/dist/nlp/deterministic-planner-suggestions.js +37 -11
- package/dist/nlp/plan-normalizer.js +18 -2
- package/dist/ops/ops-engine.js +219 -14
- package/dist/state/session-state.d.ts +56 -1
- package/dist/state/session-state.js +92 -6
- package/dist/state/sqlite-store-singleton.d.ts +22 -0
- package/dist/state/sqlite-store-singleton.js +49 -1
- package/dist/state/sqlite-store.d.ts +5 -0
- package/dist/state/sqlite-store.js +125 -2
- package/dist/telemetry/chat-telemetry.js +6 -1
- package/package.json +12 -16
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Access tokens for the orchestrator's protected surfaces.
|
|
3
|
+
*
|
|
4
|
+
* Lives in orchestrator-core because there are two surfaces to protect and
|
|
5
|
+
* they must agree on what a token is. The standalone Fastify server reads one
|
|
6
|
+
* off a `FastifyRequest`; library mode reads one off a Web `Request`. Same
|
|
7
|
+
* mint, same TTL, same in-memory table — only the carrier differs, which is
|
|
8
|
+
* why `extractAccessToken` takes either.
|
|
9
|
+
*
|
|
10
|
+
* The editor has always shipped the *client* half of this: PasswordGate reads
|
|
11
|
+
* `accessToken` off the /auth/verify response, a fetch shim attaches it as
|
|
12
|
+
* `x-access-token` on every orchestrator request, and EventSource URLs carry it
|
|
13
|
+
* as a query parameter because the browser API cannot send headers. The server
|
|
14
|
+
* half was never written — /auth/verify returned `{ok:true}` and nothing ever
|
|
15
|
+
* read the header. This module is that missing half.
|
|
16
|
+
*
|
|
17
|
+
* Tokens are opaque random strings held in memory. They do not survive a
|
|
18
|
+
* restart, which is correct for a bearer credential with no revocation list:
|
|
19
|
+
* the editor treats a 401 as "re-prompt for the password" and already wires
|
|
20
|
+
* that path, so a restart costs the user one password entry.
|
|
21
|
+
*
|
|
22
|
+
* Only the SHA-256 of a token is stored. A lookup therefore hashes the
|
|
23
|
+
* candidate and probes a Map by digest, so neither the probe nor a failure
|
|
24
|
+
* compares raw secret bytes.
|
|
25
|
+
*/
|
|
26
|
+
/** True when a password gate is configured, i.e. tokens mean something. */
|
|
27
|
+
export declare function isAccessGateEnabled(): boolean;
|
|
28
|
+
export declare function mintAccessToken(): string;
|
|
29
|
+
export declare function isValidAccessToken(token: string | undefined | null): boolean;
|
|
30
|
+
/** Drop every issued token. Used by tests; also a crude "log everyone out". */
|
|
31
|
+
export declare function revokeAllAccessTokens(): void;
|
|
32
|
+
type TokenCarrier = {
|
|
33
|
+
headers: Record<string, unknown>;
|
|
34
|
+
query?: unknown;
|
|
35
|
+
};
|
|
36
|
+
/**
|
|
37
|
+
* Pull a token off a request. Three carriers, because three transports:
|
|
38
|
+
* `x-access-token` (the editor's fetch shim), `Authorization: Bearer` (the
|
|
39
|
+
* conventional spelling, for scripts), and `?accessToken=` (EventSource, which
|
|
40
|
+
* cannot set headers at all).
|
|
41
|
+
*
|
|
42
|
+
* Accepts either a Fastify-shaped request (plain `headers` object, parsed
|
|
43
|
+
* `query`) or a Web `Request`, whose query has to be parsed off the URL.
|
|
44
|
+
*/
|
|
45
|
+
export declare function extractAccessToken(request: TokenCarrier | Request): string;
|
|
46
|
+
/**
|
|
47
|
+
* Check a submitted password against `ACCESS_PASSWORD_HASH`.
|
|
48
|
+
*
|
|
49
|
+
* Read at call time, never at module load: route modules are imported before
|
|
50
|
+
* `dotenv.config()` runs, so a hash set in `.env` is invisible to anything that
|
|
51
|
+
* captured it in a module-level `const`. That bug shipped once already — the
|
|
52
|
+
* gate reported itself disabled and `/auth/verify` accepted anything.
|
|
53
|
+
*
|
|
54
|
+
* Returns `false` when no hash is configured. Callers decide what an
|
|
55
|
+
* unconfigured gate means; it is not this function's business to invent a pass.
|
|
56
|
+
*/
|
|
57
|
+
export declare function verifyAccessPassword(password: string | undefined | null): boolean;
|
|
58
|
+
export {};
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Access tokens for the orchestrator's protected surfaces.
|
|
3
|
+
*
|
|
4
|
+
* Lives in orchestrator-core because there are two surfaces to protect and
|
|
5
|
+
* they must agree on what a token is. The standalone Fastify server reads one
|
|
6
|
+
* off a `FastifyRequest`; library mode reads one off a Web `Request`. Same
|
|
7
|
+
* mint, same TTL, same in-memory table — only the carrier differs, which is
|
|
8
|
+
* why `extractAccessToken` takes either.
|
|
9
|
+
*
|
|
10
|
+
* The editor has always shipped the *client* half of this: PasswordGate reads
|
|
11
|
+
* `accessToken` off the /auth/verify response, a fetch shim attaches it as
|
|
12
|
+
* `x-access-token` on every orchestrator request, and EventSource URLs carry it
|
|
13
|
+
* as a query parameter because the browser API cannot send headers. The server
|
|
14
|
+
* half was never written — /auth/verify returned `{ok:true}` and nothing ever
|
|
15
|
+
* read the header. This module is that missing half.
|
|
16
|
+
*
|
|
17
|
+
* Tokens are opaque random strings held in memory. They do not survive a
|
|
18
|
+
* restart, which is correct for a bearer credential with no revocation list:
|
|
19
|
+
* the editor treats a 401 as "re-prompt for the password" and already wires
|
|
20
|
+
* that path, so a restart costs the user one password entry.
|
|
21
|
+
*
|
|
22
|
+
* Only the SHA-256 of a token is stored. A lookup therefore hashes the
|
|
23
|
+
* candidate and probes a Map by digest, so neither the probe nor a failure
|
|
24
|
+
* compares raw secret bytes.
|
|
25
|
+
*/
|
|
26
|
+
import { createHash, randomBytes, timingSafeEqual } from "node:crypto";
|
|
27
|
+
/** How long a minted token stays valid. */
|
|
28
|
+
const TOKEN_TTL_MS = 12 * 60 * 60 * 1000;
|
|
29
|
+
/** Swept lazily on each mint; the map only ever holds one operator's sessions. */
|
|
30
|
+
const issued = new Map();
|
|
31
|
+
function digest(token) {
|
|
32
|
+
return createHash("sha256").update(token).digest("hex");
|
|
33
|
+
}
|
|
34
|
+
function sweep(now) {
|
|
35
|
+
for (const [hash, expiresAt] of issued) {
|
|
36
|
+
if (expiresAt <= now)
|
|
37
|
+
issued.delete(hash);
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
/** True when a password gate is configured, i.e. tokens mean something. */
|
|
41
|
+
export function isAccessGateEnabled() {
|
|
42
|
+
return Boolean(process.env.ACCESS_PASSWORD_HASH?.trim());
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* A static token for non-browser callers (CI, scripts, an MCP client) that have
|
|
46
|
+
* no way to complete a password exchange. Optional; unset means browser-only.
|
|
47
|
+
*/
|
|
48
|
+
function staticToken() {
|
|
49
|
+
return process.env.ORCHESTRATOR_ACCESS_TOKEN?.trim() ?? "";
|
|
50
|
+
}
|
|
51
|
+
export function mintAccessToken() {
|
|
52
|
+
const now = Date.now();
|
|
53
|
+
sweep(now);
|
|
54
|
+
const token = randomBytes(32).toString("base64url");
|
|
55
|
+
issued.set(digest(token), now + TOKEN_TTL_MS);
|
|
56
|
+
return token;
|
|
57
|
+
}
|
|
58
|
+
/** Constant-time equality for two strings of any length. */
|
|
59
|
+
function safeEqual(a, b) {
|
|
60
|
+
const ab = Buffer.from(a);
|
|
61
|
+
const bb = Buffer.from(b);
|
|
62
|
+
// timingSafeEqual throws on a length mismatch, which would itself leak the
|
|
63
|
+
// length. Compare digests instead: always 32 bytes, whatever the inputs.
|
|
64
|
+
const ah = createHash("sha256").update(ab).digest();
|
|
65
|
+
const bh = createHash("sha256").update(bb).digest();
|
|
66
|
+
return timingSafeEqual(ah, bh);
|
|
67
|
+
}
|
|
68
|
+
export function isValidAccessToken(token) {
|
|
69
|
+
const candidate = token?.trim();
|
|
70
|
+
if (!candidate)
|
|
71
|
+
return false;
|
|
72
|
+
const fixed = staticToken();
|
|
73
|
+
if (fixed && safeEqual(candidate, fixed))
|
|
74
|
+
return true;
|
|
75
|
+
const expiresAt = issued.get(digest(candidate));
|
|
76
|
+
if (expiresAt === undefined)
|
|
77
|
+
return false;
|
|
78
|
+
if (expiresAt <= Date.now()) {
|
|
79
|
+
issued.delete(digest(candidate));
|
|
80
|
+
return false;
|
|
81
|
+
}
|
|
82
|
+
return true;
|
|
83
|
+
}
|
|
84
|
+
/** Drop every issued token. Used by tests; also a crude "log everyone out". */
|
|
85
|
+
export function revokeAllAccessTokens() {
|
|
86
|
+
issued.clear();
|
|
87
|
+
}
|
|
88
|
+
function isWebRequest(value) {
|
|
89
|
+
return typeof value.url === "string" && typeof value.headers === "object"
|
|
90
|
+
&& typeof value.headers?.get === "function";
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Pull a token off a request. Three carriers, because three transports:
|
|
94
|
+
* `x-access-token` (the editor's fetch shim), `Authorization: Bearer` (the
|
|
95
|
+
* conventional spelling, for scripts), and `?accessToken=` (EventSource, which
|
|
96
|
+
* cannot set headers at all).
|
|
97
|
+
*
|
|
98
|
+
* Accepts either a Fastify-shaped request (plain `headers` object, parsed
|
|
99
|
+
* `query`) or a Web `Request`, whose query has to be parsed off the URL.
|
|
100
|
+
*/
|
|
101
|
+
export function extractAccessToken(request) {
|
|
102
|
+
if (isWebRequest(request)) {
|
|
103
|
+
const header = request.headers.get("x-access-token");
|
|
104
|
+
if (header?.trim())
|
|
105
|
+
return header.trim();
|
|
106
|
+
const auth = request.headers.get("authorization");
|
|
107
|
+
if (auth) {
|
|
108
|
+
const match = /^Bearer\s+(.+)$/i.exec(auth.trim());
|
|
109
|
+
if (match?.[1])
|
|
110
|
+
return match[1].trim();
|
|
111
|
+
}
|
|
112
|
+
try {
|
|
113
|
+
const fromQuery = new URL(request.url).searchParams.get("accessToken");
|
|
114
|
+
if (fromQuery?.trim())
|
|
115
|
+
return fromQuery.trim();
|
|
116
|
+
}
|
|
117
|
+
catch { /* a relative URL cannot carry a query we can read */ }
|
|
118
|
+
return "";
|
|
119
|
+
}
|
|
120
|
+
const header = request.headers["x-access-token"];
|
|
121
|
+
if (typeof header === "string" && header.trim())
|
|
122
|
+
return header.trim();
|
|
123
|
+
const auth = request.headers.authorization;
|
|
124
|
+
if (typeof auth === "string") {
|
|
125
|
+
const match = /^Bearer\s+(.+)$/i.exec(auth.trim());
|
|
126
|
+
if (match?.[1])
|
|
127
|
+
return match[1].trim();
|
|
128
|
+
}
|
|
129
|
+
const query = request.query;
|
|
130
|
+
if (query && typeof query === "object") {
|
|
131
|
+
const fromQuery = query.accessToken;
|
|
132
|
+
if (typeof fromQuery === "string" && fromQuery.trim())
|
|
133
|
+
return fromQuery.trim();
|
|
134
|
+
}
|
|
135
|
+
return "";
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* Check a submitted password against `ACCESS_PASSWORD_HASH`.
|
|
139
|
+
*
|
|
140
|
+
* Read at call time, never at module load: route modules are imported before
|
|
141
|
+
* `dotenv.config()` runs, so a hash set in `.env` is invisible to anything that
|
|
142
|
+
* captured it in a module-level `const`. That bug shipped once already — the
|
|
143
|
+
* gate reported itself disabled and `/auth/verify` accepted anything.
|
|
144
|
+
*
|
|
145
|
+
* Returns `false` when no hash is configured. Callers decide what an
|
|
146
|
+
* unconfigured gate means; it is not this function's business to invent a pass.
|
|
147
|
+
*/
|
|
148
|
+
export function verifyAccessPassword(password) {
|
|
149
|
+
const configuredHash = process.env.ACCESS_PASSWORD_HASH?.trim();
|
|
150
|
+
if (!configuredHash)
|
|
151
|
+
return false;
|
|
152
|
+
const candidate = password?.trim();
|
|
153
|
+
if (!candidate)
|
|
154
|
+
return false;
|
|
155
|
+
const hash = createHash("sha256").update(candidate).digest("hex");
|
|
156
|
+
const ab = Buffer.from(hash, "hex");
|
|
157
|
+
const bb = Buffer.from(configuredHash, "hex");
|
|
158
|
+
if (ab.length !== bb.length || ab.length === 0)
|
|
159
|
+
return false;
|
|
160
|
+
return timingSafeEqual(ab, bb);
|
|
161
|
+
}
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Voice-input transcription, as a transport-agnostic action.
|
|
3
|
+
*
|
|
4
|
+
* It lived only in `apps/orchestrator/src/routes/media.ts`, wired straight into
|
|
5
|
+
* Fastify. Library mode — `createOrchestrator()` in the site SDK — serves the
|
|
6
|
+
* same editor from a plain `Request`/`Response` handler and reimplements
|
|
7
|
+
* whichever routes somebody remembered to add. It never had this one, so a host
|
|
8
|
+
* that had funded an `OPENAI_API_KEY` still got a mic button that answered "not
|
|
9
|
+
* handled by createOrchestrator()" — the editor shows the button off the very
|
|
10
|
+
* same key that this route reads.
|
|
11
|
+
*
|
|
12
|
+
* The route is entirely session-free: no SQLite, no runtime, nothing from
|
|
13
|
+
* `session-state.js`. Its only coupling to Fastify was multipart decoding, so
|
|
14
|
+
* the action takes bytes that the transport has already decoded and both sides
|
|
15
|
+
* keep their own reader — Fastify streams the part (aborting mid-upload once it
|
|
16
|
+
* crosses the cap), library mode reads a `File` off `request.formData()`.
|
|
17
|
+
*/
|
|
18
|
+
import type { ActionResult } from "./history-actions.js";
|
|
19
|
+
export type { ActionResult };
|
|
20
|
+
/**
|
|
21
|
+
* Formats both providers read natively.
|
|
22
|
+
*
|
|
23
|
+
* `audio/ogg` is here because Firefox's MediaRecorder emits it and nothing else;
|
|
24
|
+
* dropping it would silently disable voice input on one browser.
|
|
25
|
+
*/
|
|
26
|
+
export declare const TRANSCRIPTION_MIME_TYPES: ReadonlySet<string>;
|
|
27
|
+
/** OpenAI's own per-file ceiling for the transcription endpoint. */
|
|
28
|
+
export declare const MAX_TRANSCRIPTION_BYTES: number;
|
|
29
|
+
/** One decoded upload. `filename` is optional because only OpenAI's SDK wants one. */
|
|
30
|
+
export type AudioInput = {
|
|
31
|
+
bytes: Uint8Array;
|
|
32
|
+
mimeType: string;
|
|
33
|
+
filename?: string;
|
|
34
|
+
};
|
|
35
|
+
export type TranscriptionBody = {
|
|
36
|
+
text: string;
|
|
37
|
+
provider: "openai" | "gemini";
|
|
38
|
+
model: string;
|
|
39
|
+
bytes: number;
|
|
40
|
+
mimeType: string;
|
|
41
|
+
/** True when the transcript did not come from the first thing we tried. */
|
|
42
|
+
fallbackUsed: boolean;
|
|
43
|
+
};
|
|
44
|
+
/**
|
|
45
|
+
* Seams for tests. Both providers are reached by network on every real call, so
|
|
46
|
+
* without these the action is only testable down to its validation branches.
|
|
47
|
+
*/
|
|
48
|
+
export type TranscribeDeps = {
|
|
49
|
+
transcribeWithOpenAI?: (args: {
|
|
50
|
+
input: AudioInput;
|
|
51
|
+
filename: string;
|
|
52
|
+
model: string;
|
|
53
|
+
}) => Promise<string>;
|
|
54
|
+
transcribeWithGemini?: (args: {
|
|
55
|
+
input: AudioInput;
|
|
56
|
+
model: string;
|
|
57
|
+
}) => Promise<string>;
|
|
58
|
+
};
|
|
59
|
+
/** Whether either provider key is present — the same test `/status/planner` reports as `features.audioTranscription`. */
|
|
60
|
+
export declare function isTranscriptionConfigured(): boolean;
|
|
61
|
+
/**
|
|
62
|
+
* The 503 to send when neither key is set, or `null` when at least one provider
|
|
63
|
+
* is usable. Callers check this before touching the body: an unconfigured
|
|
64
|
+
* deployment should not be made to upload 25MB to learn that.
|
|
65
|
+
*/
|
|
66
|
+
export declare function transcriptionUnavailable(): ActionResult | null;
|
|
67
|
+
/**
|
|
68
|
+
* `audio/webm;codecs=opus` → `audio/webm`.
|
|
69
|
+
*
|
|
70
|
+
* The two transports disagree about this and only one of them is visible in a
|
|
71
|
+
* test. `MediaRecorder` reports its type *with* the codec parameter, and the
|
|
72
|
+
* editor puts that string on the `File` it uploads. Busboy hands Fastify the
|
|
73
|
+
* bare `type/subtype`, so the standalone server never saw the parameter and the
|
|
74
|
+
* exact-set check below has always passed there — while `request.formData()`
|
|
75
|
+
* preserves the whole header value, so library mode would have answered 415 to
|
|
76
|
+
* every recording Chrome makes.
|
|
77
|
+
*
|
|
78
|
+
* Normalizing here rather than at each call site is deliberate: a caller that
|
|
79
|
+
* forgets is not wrong at compile time and not wrong on Firefox, only on the
|
|
80
|
+
* browser most people use. The normalized form is also what goes to the
|
|
81
|
+
* providers and comes back in the response body, so both transports report the
|
|
82
|
+
* same `mimeType` for the same recording.
|
|
83
|
+
*/
|
|
84
|
+
export declare function normalizeAudioMimeType(raw: string): string;
|
|
85
|
+
/**
|
|
86
|
+
* The half of the check that needs no bytes, split out for streaming callers.
|
|
87
|
+
*
|
|
88
|
+
* A transport that reads the part incrementally knows the MIME type from the
|
|
89
|
+
* part headers and wants to refuse an unsupported one before pulling 25MB off
|
|
90
|
+
* the socket. Without this it would have to either buffer first or write its
|
|
91
|
+
* own 415 — and a second copy of that body is how the two hosts start
|
|
92
|
+
* answering the same bad upload differently.
|
|
93
|
+
*/
|
|
94
|
+
export declare function validateAudioMimeType(mimeType: string): ActionResult | null;
|
|
95
|
+
/**
|
|
96
|
+
* Reject an upload we cannot transcribe, or `null` when it is acceptable.
|
|
97
|
+
*
|
|
98
|
+
* The order matters and mirrors what a streaming transport can actually check:
|
|
99
|
+
* the MIME type is known from the part headers before a single byte is read, so
|
|
100
|
+
* it is rejected first; size and emptiness are only knowable afterwards. A
|
|
101
|
+
* transport that enforces the cap while streaming will never see the 413 here,
|
|
102
|
+
* but a caller that buffers first (library mode, an MCP tool, a test) gets the
|
|
103
|
+
* same answer from one place instead of inventing its own.
|
|
104
|
+
*/
|
|
105
|
+
export declare function validateAudioInput(input: {
|
|
106
|
+
mimeType: string;
|
|
107
|
+
byteLength: number;
|
|
108
|
+
}): ActionResult | null;
|
|
109
|
+
/** Comma-separated env lists, tolerant of stray whitespace and trailing commas. */
|
|
110
|
+
export declare function parseTranscriptionModelList(raw: string | undefined): string[];
|
|
111
|
+
export declare function getGeminiTranscribeModel(): string;
|
|
112
|
+
/**
|
|
113
|
+
* Transcribe already-decoded audio, OpenAI first and Gemini second.
|
|
114
|
+
*
|
|
115
|
+
* Every attempt that fails contributes a line to `errors` rather than ending
|
|
116
|
+
* the call, because the whole point of the ladder is that one dead provider
|
|
117
|
+
* must not cost the user their recording. Only when nothing worked do those
|
|
118
|
+
* lines come back as the 502 detail — that string is the only diagnostic a
|
|
119
|
+
* hosted deployment gets, so it is worth keeping specific.
|
|
120
|
+
*/
|
|
121
|
+
export declare function transcribeAudioAction(rawInput: AudioInput, deps?: TranscribeDeps): Promise<ActionResult>;
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Voice-input transcription, as a transport-agnostic action.
|
|
3
|
+
*
|
|
4
|
+
* It lived only in `apps/orchestrator/src/routes/media.ts`, wired straight into
|
|
5
|
+
* Fastify. Library mode — `createOrchestrator()` in the site SDK — serves the
|
|
6
|
+
* same editor from a plain `Request`/`Response` handler and reimplements
|
|
7
|
+
* whichever routes somebody remembered to add. It never had this one, so a host
|
|
8
|
+
* that had funded an `OPENAI_API_KEY` still got a mic button that answered "not
|
|
9
|
+
* handled by createOrchestrator()" — the editor shows the button off the very
|
|
10
|
+
* same key that this route reads.
|
|
11
|
+
*
|
|
12
|
+
* The route is entirely session-free: no SQLite, no runtime, nothing from
|
|
13
|
+
* `session-state.js`. Its only coupling to Fastify was multipart decoding, so
|
|
14
|
+
* the action takes bytes that the transport has already decoded and both sides
|
|
15
|
+
* keep their own reader — Fastify streams the part (aborting mid-upload once it
|
|
16
|
+
* crosses the cap), library mode reads a `File` off `request.formData()`.
|
|
17
|
+
*/
|
|
18
|
+
import OpenAI from "openai";
|
|
19
|
+
import { toFile } from "openai/uploads";
|
|
20
|
+
import { toErrorDetail } from "../errors.js";
|
|
21
|
+
import { getGeminiClient } from "../image/image-helpers.js";
|
|
22
|
+
/**
|
|
23
|
+
* Formats both providers read natively.
|
|
24
|
+
*
|
|
25
|
+
* `audio/ogg` is here because Firefox's MediaRecorder emits it and nothing else;
|
|
26
|
+
* dropping it would silently disable voice input on one browser.
|
|
27
|
+
*/
|
|
28
|
+
export const TRANSCRIPTION_MIME_TYPES = new Set([
|
|
29
|
+
"audio/mp3",
|
|
30
|
+
"audio/mpeg",
|
|
31
|
+
"audio/mp4",
|
|
32
|
+
"audio/mpga",
|
|
33
|
+
"audio/m4a",
|
|
34
|
+
"audio/wav",
|
|
35
|
+
"audio/webm",
|
|
36
|
+
"audio/ogg"
|
|
37
|
+
]);
|
|
38
|
+
/** OpenAI's own per-file ceiling for the transcription endpoint. */
|
|
39
|
+
export const MAX_TRANSCRIPTION_BYTES = 25 * 1024 * 1024;
|
|
40
|
+
/** Whether either provider key is present — the same test `/status/planner` reports as `features.audioTranscription`. */
|
|
41
|
+
export function isTranscriptionConfigured() {
|
|
42
|
+
return Boolean(process.env.OPENAI_API_KEY?.trim() || process.env.GOOGLE_GENAI_API_KEY?.trim());
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* The 503 to send when neither key is set, or `null` when at least one provider
|
|
46
|
+
* is usable. Callers check this before touching the body: an unconfigured
|
|
47
|
+
* deployment should not be made to upload 25MB to learn that.
|
|
48
|
+
*/
|
|
49
|
+
export function transcriptionUnavailable() {
|
|
50
|
+
if (isTranscriptionConfigured())
|
|
51
|
+
return null;
|
|
52
|
+
return {
|
|
53
|
+
code: 503,
|
|
54
|
+
body: { error: "no transcription provider configured (set OPENAI_API_KEY or GOOGLE_GENAI_API_KEY)" }
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* `audio/webm;codecs=opus` → `audio/webm`.
|
|
59
|
+
*
|
|
60
|
+
* The two transports disagree about this and only one of them is visible in a
|
|
61
|
+
* test. `MediaRecorder` reports its type *with* the codec parameter, and the
|
|
62
|
+
* editor puts that string on the `File` it uploads. Busboy hands Fastify the
|
|
63
|
+
* bare `type/subtype`, so the standalone server never saw the parameter and the
|
|
64
|
+
* exact-set check below has always passed there — while `request.formData()`
|
|
65
|
+
* preserves the whole header value, so library mode would have answered 415 to
|
|
66
|
+
* every recording Chrome makes.
|
|
67
|
+
*
|
|
68
|
+
* Normalizing here rather than at each call site is deliberate: a caller that
|
|
69
|
+
* forgets is not wrong at compile time and not wrong on Firefox, only on the
|
|
70
|
+
* browser most people use. The normalized form is also what goes to the
|
|
71
|
+
* providers and comes back in the response body, so both transports report the
|
|
72
|
+
* same `mimeType` for the same recording.
|
|
73
|
+
*/
|
|
74
|
+
export function normalizeAudioMimeType(raw) {
|
|
75
|
+
return (raw.split(";")[0] ?? "").trim().toLowerCase();
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* The half of the check that needs no bytes, split out for streaming callers.
|
|
79
|
+
*
|
|
80
|
+
* A transport that reads the part incrementally knows the MIME type from the
|
|
81
|
+
* part headers and wants to refuse an unsupported one before pulling 25MB off
|
|
82
|
+
* the socket. Without this it would have to either buffer first or write its
|
|
83
|
+
* own 415 — and a second copy of that body is how the two hosts start
|
|
84
|
+
* answering the same bad upload differently.
|
|
85
|
+
*/
|
|
86
|
+
export function validateAudioMimeType(mimeType) {
|
|
87
|
+
if (!TRANSCRIPTION_MIME_TYPES.has(normalizeAudioMimeType(mimeType))) {
|
|
88
|
+
// The raw value is echoed, not the normalized one — someone reading this
|
|
89
|
+
// error wants to see what their client actually sent.
|
|
90
|
+
return { code: 415, body: { error: `unsupported audio type: ${mimeType}` } };
|
|
91
|
+
}
|
|
92
|
+
return null;
|
|
93
|
+
}
|
|
94
|
+
/**
|
|
95
|
+
* Reject an upload we cannot transcribe, or `null` when it is acceptable.
|
|
96
|
+
*
|
|
97
|
+
* The order matters and mirrors what a streaming transport can actually check:
|
|
98
|
+
* the MIME type is known from the part headers before a single byte is read, so
|
|
99
|
+
* it is rejected first; size and emptiness are only knowable afterwards. A
|
|
100
|
+
* transport that enforces the cap while streaming will never see the 413 here,
|
|
101
|
+
* but a caller that buffers first (library mode, an MCP tool, a test) gets the
|
|
102
|
+
* same answer from one place instead of inventing its own.
|
|
103
|
+
*/
|
|
104
|
+
export function validateAudioInput(input) {
|
|
105
|
+
const badType = validateAudioMimeType(input.mimeType);
|
|
106
|
+
if (badType)
|
|
107
|
+
return badType;
|
|
108
|
+
if (input.byteLength > MAX_TRANSCRIPTION_BYTES) {
|
|
109
|
+
return { code: 413, body: { error: "audio file is too large (max 25MB)" } };
|
|
110
|
+
}
|
|
111
|
+
if (input.byteLength === 0) {
|
|
112
|
+
return { code: 400, body: { error: "audio file is empty" } };
|
|
113
|
+
}
|
|
114
|
+
return null;
|
|
115
|
+
}
|
|
116
|
+
/** Comma-separated env lists, tolerant of stray whitespace and trailing commas. */
|
|
117
|
+
export function parseTranscriptionModelList(raw) {
|
|
118
|
+
if (!raw)
|
|
119
|
+
return [];
|
|
120
|
+
return raw
|
|
121
|
+
.split(",")
|
|
122
|
+
.map((item) => item.trim())
|
|
123
|
+
.filter((item) => item.length > 0);
|
|
124
|
+
}
|
|
125
|
+
export function getGeminiTranscribeModel() {
|
|
126
|
+
return process.env.GOOGLE_GENAI_TRANSCRIBE_MODEL?.trim() || "gemini-2.5-flash";
|
|
127
|
+
}
|
|
128
|
+
// A view, not a copy: the transport already allocated these bytes and an audio
|
|
129
|
+
// upload can be 25MB.
|
|
130
|
+
function asBuffer(bytes) {
|
|
131
|
+
return Buffer.isBuffer(bytes) ? bytes : Buffer.from(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* Gemini reads audio natively (webm/ogg/wav/mp4/m4a/mp3 all verified), so it
|
|
135
|
+
* serves as a drop-in fallback when OpenAI is unavailable — and as the only
|
|
136
|
+
* transcriber in deployments that fund a Google key and nothing else.
|
|
137
|
+
*/
|
|
138
|
+
async function geminiTranscribe(args) {
|
|
139
|
+
const ai = await getGeminiClient();
|
|
140
|
+
const response = await ai.models.generateContent({
|
|
141
|
+
model: args.model,
|
|
142
|
+
contents: [
|
|
143
|
+
{
|
|
144
|
+
role: "user",
|
|
145
|
+
parts: [
|
|
146
|
+
{
|
|
147
|
+
text: "Transcribe this audio verbatim. Return only the spoken words as plain text — no preamble, quotes, or commentary. If there is no intelligible speech, return an empty string."
|
|
148
|
+
},
|
|
149
|
+
{ inlineData: { mimeType: args.input.mimeType, data: asBuffer(args.input.bytes).toString("base64") } }
|
|
150
|
+
]
|
|
151
|
+
}
|
|
152
|
+
]
|
|
153
|
+
});
|
|
154
|
+
return (response.text ?? "").trim();
|
|
155
|
+
}
|
|
156
|
+
/**
|
|
157
|
+
* Transcribe already-decoded audio, OpenAI first and Gemini second.
|
|
158
|
+
*
|
|
159
|
+
* Every attempt that fails contributes a line to `errors` rather than ending
|
|
160
|
+
* the call, because the whole point of the ladder is that one dead provider
|
|
161
|
+
* must not cost the user their recording. Only when nothing worked do those
|
|
162
|
+
* lines come back as the 502 detail — that string is the only diagnostic a
|
|
163
|
+
* hosted deployment gets, so it is worth keeping specific.
|
|
164
|
+
*/
|
|
165
|
+
export async function transcribeAudioAction(rawInput, deps = {}) {
|
|
166
|
+
const input = { ...rawInput, mimeType: normalizeAudioMimeType(rawInput.mimeType) };
|
|
167
|
+
const hasOpenAI = Boolean(process.env.OPENAI_API_KEY?.trim());
|
|
168
|
+
const hasGemini = Boolean(process.env.GOOGLE_GENAI_API_KEY?.trim());
|
|
169
|
+
const bytes = input.bytes.byteLength;
|
|
170
|
+
const errors = [];
|
|
171
|
+
// Primary: OpenAI Whisper-family models (best quality). Try the configured
|
|
172
|
+
// model, then any fallbacks, before giving up on the provider.
|
|
173
|
+
if (hasOpenAI) {
|
|
174
|
+
const primaryModel = process.env.OPENAI_TRANSCRIBE_MODEL?.trim() || "gpt-4o-mini-transcribe";
|
|
175
|
+
const fallbackModels = parseTranscriptionModelList(process.env.OPENAI_TRANSCRIBE_FALLBACK_MODELS);
|
|
176
|
+
const modelsToTry = Array.from(new Set([primaryModel, ...fallbackModels]));
|
|
177
|
+
const filename = input.filename || `recording.${input.mimeType.split("/")[1] ?? "webm"}`;
|
|
178
|
+
try {
|
|
179
|
+
// Building the client and the uploadable file is done once, outside the
|
|
180
|
+
// model loop, and a failure here is attributed to the provider rather
|
|
181
|
+
// than to any one model — it means we never got as far as a request.
|
|
182
|
+
const transcribe = deps.transcribeWithOpenAI ?? (await createOpenAITranscriber(input, filename));
|
|
183
|
+
for (const model of modelsToTry) {
|
|
184
|
+
try {
|
|
185
|
+
const text = await transcribe({ input, filename, model });
|
|
186
|
+
return {
|
|
187
|
+
code: 200,
|
|
188
|
+
body: {
|
|
189
|
+
text,
|
|
190
|
+
provider: "openai",
|
|
191
|
+
model,
|
|
192
|
+
bytes,
|
|
193
|
+
mimeType: input.mimeType,
|
|
194
|
+
fallbackUsed: model !== primaryModel
|
|
195
|
+
}
|
|
196
|
+
};
|
|
197
|
+
}
|
|
198
|
+
catch (error) {
|
|
199
|
+
errors.push(`openai/${model}: ${toErrorDetail(error)}`);
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
catch (error) {
|
|
204
|
+
errors.push(`openai: ${toErrorDetail(error)}`);
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
if (hasGemini) {
|
|
208
|
+
const model = getGeminiTranscribeModel();
|
|
209
|
+
try {
|
|
210
|
+
const text = await (deps.transcribeWithGemini ?? geminiTranscribe)({ input, model });
|
|
211
|
+
return {
|
|
212
|
+
code: 200,
|
|
213
|
+
body: {
|
|
214
|
+
text,
|
|
215
|
+
provider: "gemini",
|
|
216
|
+
model,
|
|
217
|
+
bytes,
|
|
218
|
+
mimeType: input.mimeType,
|
|
219
|
+
// Gemini is the fallback only if OpenAI was configured and lost; on a
|
|
220
|
+
// Google-only deployment it is the primary.
|
|
221
|
+
fallbackUsed: hasOpenAI
|
|
222
|
+
}
|
|
223
|
+
};
|
|
224
|
+
}
|
|
225
|
+
catch (error) {
|
|
226
|
+
errors.push(`gemini: ${toErrorDetail(error)}`);
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
return {
|
|
230
|
+
code: 502,
|
|
231
|
+
body: { error: "transcription failed", detail: errors.join(" | ").slice(0, 1200) }
|
|
232
|
+
};
|
|
233
|
+
}
|
|
234
|
+
/**
|
|
235
|
+
* Bind an OpenAI client and one uploadable copy of the audio to a per-model call.
|
|
236
|
+
*
|
|
237
|
+
* `maxRetries: 0` and a bounded timeout are load-bearing: the SDK otherwise
|
|
238
|
+
* retries a 429 with long backoff, so an over-quota account would stall the
|
|
239
|
+
* request for minutes before the Gemini fallback ever got its turn.
|
|
240
|
+
*/
|
|
241
|
+
async function createOpenAITranscriber(input, filename) {
|
|
242
|
+
const client = new OpenAI({ apiKey: process.env.OPENAI_API_KEY, maxRetries: 0, timeout: 20000 });
|
|
243
|
+
const audioFile = await toFile(asBuffer(input.bytes), filename, { type: input.mimeType });
|
|
244
|
+
return async ({ model }) => {
|
|
245
|
+
const transcription = await client.audio.transcriptions.create({ file: audioFile, model });
|
|
246
|
+
return transcription.text ?? "";
|
|
247
|
+
};
|
|
248
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Block discovery, as a transport-agnostic action.
|
|
3
|
+
*
|
|
4
|
+
* The MCP server's `avocado-list-block-types` and `avocado-get-block-schema`
|
|
5
|
+
* used to read the block registry of their *own* process. That process never
|
|
6
|
+
* calls the target site's `registerBlocks()`, so for any site with custom
|
|
7
|
+
* blocks the answer was not merely incomplete, it was inverted: the agent was
|
|
8
|
+
* handed Avocado's twenty built-ins — types the site cannot render — and told
|
|
9
|
+
* that the site's own block types were unknown. Discovery is an agent's first
|
|
10
|
+
* move against a new site, so it planned every subsequent `add_block` from a
|
|
11
|
+
* confident, complete, wrong answer.
|
|
12
|
+
*
|
|
13
|
+
* The registry that matters is the one in the process that will render the
|
|
14
|
+
* page. In library mode that is this process: `createOrchestrator()` runs
|
|
15
|
+
* inside the host's Next app and calls its `registerBlocks()`. Serving the
|
|
16
|
+
* manifest from here therefore answers with the site's real vocabulary, and
|
|
17
|
+
* the MCP server can stop guessing from module scope.
|
|
18
|
+
*/
|
|
19
|
+
export type ActionResult = {
|
|
20
|
+
code: number;
|
|
21
|
+
body: unknown;
|
|
22
|
+
};
|
|
23
|
+
/**
|
|
24
|
+
* GET /blocks/manifest — every block type registered in this process.
|
|
25
|
+
*
|
|
26
|
+
* Deliberately unscoped: there is no `session` or `siteId` parameter, because
|
|
27
|
+
* the block vocabulary is a property of the running process, not of a draft.
|
|
28
|
+
* The standalone multi-site server therefore answers with its built-ins for
|
|
29
|
+
* every site it hosts, which is exactly what it can actually render.
|
|
30
|
+
*/
|
|
31
|
+
export declare function blocksManifestAction(): ActionResult;
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Block discovery, as a transport-agnostic action.
|
|
3
|
+
*
|
|
4
|
+
* The MCP server's `avocado-list-block-types` and `avocado-get-block-schema`
|
|
5
|
+
* used to read the block registry of their *own* process. That process never
|
|
6
|
+
* calls the target site's `registerBlocks()`, so for any site with custom
|
|
7
|
+
* blocks the answer was not merely incomplete, it was inverted: the agent was
|
|
8
|
+
* handed Avocado's twenty built-ins — types the site cannot render — and told
|
|
9
|
+
* that the site's own block types were unknown. Discovery is an agent's first
|
|
10
|
+
* move against a new site, so it planned every subsequent `add_block` from a
|
|
11
|
+
* confident, complete, wrong answer.
|
|
12
|
+
*
|
|
13
|
+
* The registry that matters is the one in the process that will render the
|
|
14
|
+
* page. In library mode that is this process: `createOrchestrator()` runs
|
|
15
|
+
* inside the host's Next app and calls its `registerBlocks()`. Serving the
|
|
16
|
+
* manifest from here therefore answers with the site's real vocabulary, and
|
|
17
|
+
* the MCP server can stop guessing from module scope.
|
|
18
|
+
*/
|
|
19
|
+
import { buildBlockManifest } from "@avocadostudio-ai/shared";
|
|
20
|
+
/**
|
|
21
|
+
* GET /blocks/manifest — every block type registered in this process.
|
|
22
|
+
*
|
|
23
|
+
* Deliberately unscoped: there is no `session` or `siteId` parameter, because
|
|
24
|
+
* the block vocabulary is a property of the running process, not of a draft.
|
|
25
|
+
* The standalone multi-site server therefore answers with its built-ins for
|
|
26
|
+
* every site it hosts, which is exactly what it can actually render.
|
|
27
|
+
*/
|
|
28
|
+
export function blocksManifestAction() {
|
|
29
|
+
const manifest = buildBlockManifest();
|
|
30
|
+
return { code: 200, body: manifest };
|
|
31
|
+
}
|