@volter/twin-fal 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,112 @@
1
+ // The fal pack's HALF of the scenario system, on the kernel's ONE engine (@volter/world-core
2
+ // scenario.ts). The GRAMMAR (ordering, once/scope/phase, extractors, strict parsing, miss
3
+ // records, twin.use) is the kernel's and identical for every vendor; this module declares only
4
+ // the VOCABULARY:
5
+ // • the `on` keys a handler may match (model id, submitted input text, the queue-vs-sync
6
+ // surface, and the 1-based ordinal of the submit within the world),
7
+ // • what a `respond` payload may contain ({ output, inferenceTime? }),
8
+ // • nothing else — fal's PROTOCOL envelope (the IN_QUEUE → IN_PROGRESS → COMPLETED machine,
9
+ // queue_position, logs, urls, metrics) stays the twin's, faithful and unscriptable. What a
10
+ // handler replaces is exactly the GENERATIVE STUB: the `output` a request settles to,
11
+ // which without a handler is the labeled `[twin-stub:fal:<hash8>]` value.
12
+ //
13
+ // Matching happens at SUBMIT (POST /{model_id}, both the queue and the fal.run sync surface) and
14
+ // the outcome is PERSISTED on the request resource, so completion stays a pure fold of the kernel
15
+ // log — the status progression never consults the engine and holds no hidden session state.
16
+ // `nthRequest` is a PACK matcher over the count of request rows already in the root, not the
17
+ // engine's call counter.
18
+ //
19
+ // The handler FILE (handlers/fal.json in a world dir) is the only write surface.
20
+ import { getActiveWorldStore, parseScenarioDocument, type PackScenarioAdapter, ScenarioError, type ScenarioDocument, ScenarioEngine, type ScenarioFeatures } from '@volter/world-core';
21
+
22
+ /** The request slice the scenario system sees — built by the submit route from the parsed body. */
23
+ export type FalScenarioRequest = {
24
+ modelId: string;
25
+ input: Record<string, unknown>;
26
+ /** The surface the submit arrived on: fal's queue (queue.fal.run) or the inline fal.run. */
27
+ surface: 'queue' | 'sync';
28
+ /** 1-based ordinal of this submit among the request rows in the root. */
29
+ nthRequest: number;
30
+ };
31
+ export type FalScenarioEngine = ScenarioEngine<FalScenarioRequest>;
32
+
33
+ /** The scripted outcome: the `output` the request settles to (any JSON value the model would
34
+ * return — real fal models answer objects like `{ images: [...] }`, not only strings), and
35
+ * optionally the reported inference time. */
36
+ export type FalScenarioRespond = {
37
+ output: unknown;
38
+ inferenceTime?: number;
39
+ };
40
+
41
+ const RESPOND_KEYS = new Set(['output', 'inferenceTime']);
42
+ const nonEmptyString = (cond: unknown): cond is string => typeof cond === 'string' && cond.length > 0;
43
+
44
+ /** Stable text of a submitted input — the corpus `inputIncludes` and `textPattern` extractors
45
+ * read. Key order is sorted so the same input always yields the same corpus. */
46
+ export function canonicalInputText(input: unknown): string {
47
+ if (input === null || typeof input !== 'object') return JSON.stringify(input) ?? 'null';
48
+ if (Array.isArray(input)) return `[${input.map(canonicalInputText).join(',')}]`;
49
+ const keys = Object.keys(input as Record<string, unknown>).sort();
50
+ return `{${keys.map((k) => `${JSON.stringify(k)}:${canonicalInputText((input as Record<string, unknown>)[k])}`).join(',')}}`;
51
+ }
52
+
53
+ export const falScenarioAdapter: PackScenarioAdapter<FalScenarioRequest> = {
54
+ vendor: 'fal',
55
+ features: (req): ScenarioFeatures => ({
56
+ modelId: req.modelId,
57
+ input: canonicalInputText(req.input).slice(0, 300),
58
+ surface: req.surface,
59
+ nthRequest: req.nthRequest,
60
+ }),
61
+ matchers: {
62
+ modelIdEquals: (req, cond) => nonEmptyString(cond) && req.modelId === cond,
63
+ modelIdIncludes: (req, cond) => nonEmptyString(cond) && req.modelId.toLowerCase().includes(cond.toLowerCase()),
64
+ inputIncludes: (req, cond) => nonEmptyString(cond) && canonicalInputText(req.input).toLowerCase().includes(cond.toLowerCase()),
65
+ surfaceEquals: (req, cond) => (cond === 'queue' || cond === 'sync') && req.surface === cond,
66
+ // The 1-based index among requests submitted into the ROOT, not the engine's call counter —
67
+ // no hidden session state.
68
+ nthRequest: (req, cond) => Number.isInteger(cond) && req.nthRequest === cond,
69
+ },
70
+ text: (req) => canonicalInputText(req.input),
71
+ validateOn: (on) => {
72
+ for (const k of ['modelIdEquals', 'modelIdIncludes', 'inputIncludes'] as const) {
73
+ if (on[k] !== undefined && (typeof on[k] !== 'string' || !on[k])) return `on.${k} is a non-empty string`;
74
+ }
75
+ if (on.surfaceEquals !== undefined && on.surfaceEquals !== 'queue' && on.surfaceEquals !== 'sync') return 'on.surfaceEquals is "queue" or "sync"';
76
+ if (on.nthRequest !== undefined && (!Number.isInteger(on.nthRequest) || (on.nthRequest as number) < 1)) return 'on.nthRequest is a positive integer';
77
+ return null;
78
+ },
79
+ validateRespond: (respond) => {
80
+ if (typeof respond !== 'object' || respond === null || Array.isArray(respond)) return 'respond is an object { output, inferenceTime? }';
81
+ const r = respond as Record<string, unknown>;
82
+ for (const k of Object.keys(r)) if (!RESPOND_KEYS.has(k)) return `respond: unknown key "${k}" (valid: ${[...RESPOND_KEYS].join(', ')})`;
83
+ if (!('output' in r)) return 'respond needs output (the value the request settles to — any JSON the model would return)';
84
+ if (r.output === undefined) return 'respond.output is a JSON value, not undefined';
85
+ if (r.inferenceTime !== undefined && (typeof r.inferenceTime !== 'number' || !Number.isFinite(r.inferenceTime) || r.inferenceTime < 0)) {
86
+ return 'respond.inferenceTime is a non-negative number of SECONDS';
87
+ }
88
+ return null;
89
+ },
90
+ };
91
+
92
+ /** Load + strictly validate a handlers document (handlers/fal.json). A broken file fails server
93
+ * construction loudly; it never falls back or misfires silently. */
94
+ export function loadFalScenarioDocument(path: string): ScenarioDocument {
95
+ let parsed: unknown;
96
+ try {
97
+ // Read through the ACTIVE WorldStore, never the filesystem directly (runtime contract R12b):
98
+ // the handlers document is WORLD STATE, so a MemoryWorldStore / DO-backed world serves ITS
99
+ // OWN scenario instead of whatever happens to sit on the host disk — and the serve path
100
+ // stays workerd-clean.
101
+ const raw = getActiveWorldStore().read(path);
102
+ if (raw === null) throw new Error(`ENOENT: no such file or directory, open '${path}'`);
103
+ parsed = JSON.parse(raw);
104
+ } catch (e) {
105
+ throw new ScenarioError(`fal scenario: cannot read/parse ${path}: ${e instanceof Error ? e.message : String(e)}`);
106
+ }
107
+ return parseScenarioDocument(parsed, falScenarioAdapter);
108
+ }
109
+
110
+ export function createFalScenarioEngine(document?: ScenarioDocument): FalScenarioEngine {
111
+ return new ScenarioEngine(falScenarioAdapter, document);
112
+ }
@@ -0,0 +1,64 @@
1
+ // fal twin HTTP server — serve the full fal API twin handler over HTTP so the real
2
+ // `@fal-ai/client` SDK (via a `requestMiddleware` that rewrites the URL to this server — see
3
+ // fal-sdk.integration.test.ts) works unmodified against it. Every route is plain JSON. Request
4
+ // headers are forwarded verbatim — NOTE the server deliberately does NOT thread the incoming
5
+ // HTTP `Host` header as `handleFalTwinRequest`'s `host` param: once a caller's request has been
6
+ // rewritten to point at THIS server (127.0.0.1:<port>), the literal Host header is always
7
+ // 127.0.0.1, which carries no vendor-host information. Host-split routing instead reads the
8
+ // `x-fal-target-url` header (the real fal-js proxy protocol, which carries the ORIGINAL absolute
9
+ // URL) — see fal-twin.ts's `routeFalSurface`. Writable by default; pass readOnly to reject writes
10
+ // with 405 (D3). State is the kernel projection (no side-store) — see fal-twin.ts.
11
+ //
12
+ // FETCH-FIRST (runtime contract R12b): the serve path is the plain fetch below, built from the
13
+ // kernel's ONE adaptation (`createTwinFetchFromHandler` — the keyless read doors, body read,
14
+ // header map, worldNow() stamp, JSON reply); this file contributes only VALUES. The server is one
15
+ // line of Bun.serve around that same closure.
16
+ //
17
+ // THE READ DOORS (runtime contract R5): keyless `GET /twin` (the manifest, which teaches the
18
+ // handler shape) and `GET /twin/scenario` (active handlers + match/miss counts). Both are the
19
+ // adapter's, answered before any vendor routing, and neither takes a credential.
20
+ //
21
+ // SCENARIO scripting (fal-scenario.ts): pass `scenarioPath` (or set the TWIN_FAL_SCENARIO env
22
+ // var) to script what a submitted request settles to. A malformed scenario throws at
23
+ // construction — loudly, never a silent fallback.
24
+ import { serveHttp } from '@volter/world-core';
25
+ import { handleFalTwinRequest } from './fal-twin.ts';
26
+ import { createTwinFetchFromHandler, twinManifest } from '@volter/world-core';
27
+ import { createFalScenarioEngine, loadFalScenarioDocument, type FalScenarioEngine } from './fal-scenario.ts';
28
+
29
+ /** Options every fal-twin HTTP surface needs, independent of who owns the socket. */
30
+ export interface FalTwinFetchOptions {
31
+ root?: string;
32
+ readOnly?: boolean;
33
+ /** Path to the world dir's handlers/fal.json — read ONCE, at construction, through the active
34
+ * WorldStore. Absent → no engine, and every request settles to the labeled stub. */
35
+ scenarioPath?: string;
36
+ }
37
+
38
+ export function createFalTwinFetch(options: FalTwinFetchOptions = {}): (request: Request) => Promise<Response> {
39
+ const scenarioPath = options.scenarioPath ?? process.env.TWIN_FAL_SCENARIO;
40
+ const scenarioEngine: FalScenarioEngine | undefined = scenarioPath ? createFalScenarioEngine(loadFalScenarioDocument(scenarioPath)) : undefined;
41
+ const { scenarioPath: _scenarioPath, ...handlerOptions } = options;
42
+ return createTwinFetchFromHandler(handleFalTwinRequest, {
43
+ ...handlerOptions,
44
+ manifest: () => twinManifest({
45
+ vendor: 'fal',
46
+ twinOf: 'the fal.ai inference queue',
47
+ stateSentence: 'Stateful: it stores queued inference requests and their status/log/response lifecycle. Create them through the vendor\'s OWN API (POST /{model_id} on queue.fal.run or fal.run) with any credential.',
48
+ behaviorSentence: 'The PROTOCOL envelope (IN_QUEUE → IN_PROGRESS → COMPLETED, queue_position, logs, urls, metrics) is the twin\'s and faithful. What a request SETTLES TO is scripted by MSW-shaped handlers in the world dir (handlers/fal.json): {on:{modelIdEquals|modelIdIncludes|inputIncludes|surfaceEquals|nthRequest|nthCall}, respond:{output, inferenceTime?}, once?, phase?}. Matching happens at submit and the outcome persists on the request; unmatched submits settle to the labeled [twin-stub:fal:…] value, and the misses are inspectable at GET /twin/scenario.',
49
+ exampleHandler: { on: { modelIdIncludes: 'flux' }, respond: { output: { images: [{ url: 'https://cdn.example.com/scripted.png' }] } } },
50
+ engine: scenarioEngine as never,
51
+ }),
52
+ scenarioStatus: () => (scenarioEngine ? scenarioEngine.status() : { vendor: 'fal', handlers: [], misses: 0, recentMisses: [] }),
53
+ ...(scenarioEngine ? { handlerOptions: { scenarioEngine } } : {}),
54
+ });
55
+ }
56
+
57
+ export async function createFalTwinServer(options: { root?: string; port?: number; readOnly?: boolean; scenarioPath?: string } = {}): Promise<{ port: number; stop: () => void }> {
58
+ const server = await serveHttp({
59
+ port: options.port ?? 0,
60
+ idleTimeout: 60,
61
+ fetch: createFalTwinFetch(options),
62
+ });
63
+ return { port: server.port ?? options.port ?? 0, stop: () => server.stop(true) };
64
+ }
@@ -0,0 +1,479 @@
1
+ // fal.ai API twin REQUEST HANDLER — a v1 slice of the fal.ai queue-lifecycle model-inference
2
+ // surface, backed by the event/action-log kernel (@volter/world-core). Contract:
3
+ // handleFalTwinRequest({ method, path, body, host?, headers, root, readOnly }) -> { status, body, headers }
4
+ //
5
+ // SOURCE OF TRUTH: grounded read-only against fal.ai's own published docs (fal.ai/docs/model-
6
+ // endpoints/queue, fal.ai/docs/model-endpoints/webhooks) and the fal-js client source
7
+ // (github.com/fal-ai/fal-js — libs/client/src/{config,middleware,queue,request}.ts), fetched
8
+ // during this build (see spec-sources.json). Every route/field/status below cites one of those
9
+ // two sources; a doc-UNVERIFIED shape is annotated inline (⚠) and NOT claimed beyond what the
10
+ // verify() actually proves.
11
+ //
12
+ // Facts grounded from fal.ai/docs/model-endpoints/queue:
13
+ // - submit: POST https://queue.fal.run/{model_id} -> {request_id, response_url, status_url,
14
+ // cancel_url, queue_position}. status: GET .../requests/{request_id}/status[?logs=1] ->
15
+ // {status: IN_QUEUE|IN_PROGRESS|COMPLETED, request_id, response_url, queue_position
16
+ // (IN_QUEUE only), logs (IN_PROGRESS/COMPLETED), metrics.inference_time (COMPLETED only)}.
17
+ // - cancel: PUT .../requests/{request_id}/cancel -> 202 {"status":"CANCELLATION_REQUESTED"} |
18
+ // 400 {"status":"ALREADY_COMPLETED"} | 404 {"status":"NOT_FOUND"} (NOT a {detail} envelope —
19
+ // this exact shape is doc-grounded, distinct from every other 404 in this pack).
20
+ // - result: docs literally show GET .../requests/{request_id}/response, but the REAL fal-js
21
+ // client builds the result URL WITHOUT the /response suffix (verified against the fetched
22
+ // queue.ts source: `path: /requests/${requestId}`) — a genuine doc/client conflict (⚠2 in the
23
+ // build spec). This twin serves BOTH forms so either a doc-literal caller or the real SDK
24
+ // works; the /response form is the doc's own example, still exercised, but not separately
25
+ // claimed as its own capability (`fal.queue.response_alias_route` stays `todo`).
26
+ // Facts grounded from fal.ai/docs/model-endpoints/webhooks — see fal-webhooks.ts.
27
+ // Facts grounded from fal-js middleware.ts/config.ts/request.ts — see fal-twin.ts's
28
+ // `routeFalSurface` and fal-sdk.integration.test.ts.
29
+ //
30
+ // State lives ENTIRELY in the kernel action log: writes go through `applyTwinWrite`, reads are
31
+ // the projection (`projectResources`). There is NO Map/array side-store (D1). No real fal is
32
+ // ever contacted.
33
+ //
34
+ // THE GENERATIVE STUB (D2 honesty): fal is a GENERATIVE vendor — model inference requires
35
+ // hosted GPU weights that cannot run locally, so a request's `output` is a clearly-labeled
36
+ // DETERMINISTIC stub (`[twin-stub:fal:<hash8>]`), derived from hash(model_id, canonical-
37
+ // JSON(input)) — same input twice -> identical output, different input -> different output. The
38
+ // PROTOCOL ENVELOPE (status machine, queue_position, logs, urls, metrics shape) is faithful; only
39
+ // the generated content is a stub.
40
+ //
41
+ // Honesty (D2): an unmodeled route returns a FastAPI-style `{"detail": "..."}` 404 (fal's
42
+ // gateway is FastAPI-based per its documented 422 validation-error shape), never a fabricated
43
+ // success. readOnly rejects writes with 405.
44
+ //
45
+ // Kernel SUBJECT ids are type-prefixed (`request:<uuid>`) — the PUBLIC request_id emitted to
46
+ // clients is the bare uuid. `kid()` builds the subject id; `rows()` strips the prefix back off.
47
+ import { createHash } from 'node:crypto';
48
+ import { applyTwinWrite, projectResources } from '@volter/world-core';
49
+ import { falJwks } from './fal-webhooks.ts';
50
+ import type { FalScenarioEngine, FalScenarioRespond } from './fal-scenario.ts';
51
+
52
+ const SERVICE = 'fal';
53
+
54
+ export type FalRequest = {
55
+ method: string;
56
+ path: string;
57
+ body?: string;
58
+ /** Explicit host (e.g. from an HTTP `Host` header or a direct routing decision). Takes
59
+ * priority over `x-fal-target-url` in `headers`. */
60
+ host?: string;
61
+ headers?: Record<string, string>;
62
+ occurredAt?: string;
63
+ root?: string;
64
+ readOnly?: boolean;
65
+ /** The pack's scenario engine, mounted by the fetch adapter when a world dir carries a
66
+ * handlers/fal.json. Absent → every request settles to the labeled deterministic stub,
67
+ * unchanged. Consulted at SUBMIT only; the outcome persists on the request resource. */
68
+ scenarioEngine?: FalScenarioEngine;
69
+ };
70
+ export type FalResponse = { status: number; body: unknown; headers?: Record<string, string> };
71
+
72
+ // Resource types the twin actually PROJECTS in the kernel — the honest v1 subset. `signing_key`
73
+ // is a lazily-materialized per-root singleton (the same "fold on first touch" pattern replicate's
74
+ // webhook_secret / elevenlabs' dubbing.get use); it's never itself a REST-addressable resource
75
+ // (fal has no admin surface to list/rotate keys — see README ## Coverage), only the ed25519
76
+ // keypair backing the JWKS endpoint + webhook signing.
77
+ export const FAL_RESOURCE_TYPES = ['request', 'signing_key'] as const;
78
+ export type FalResourceType = typeof FAL_RESOURCE_TYPES[number];
79
+
80
+ function lowerHeaders(h: Record<string, string> | undefined): Record<string, string> {
81
+ const out: Record<string, string> = {};
82
+ for (const [k, v] of Object.entries(h ?? {})) out[k.toLowerCase()] = v;
83
+ return out;
84
+ }
85
+
86
+ // ── host-split surface routing (build spec §1 divergence 2) ──────────────────────────────────
87
+ // fal addresses THREE distinct real hosts with overlapping method+path shapes:
88
+ // queue.fal.run — the queue lifecycle (submit/status/result/cancel)
89
+ // fal.run — the SYNCHRONOUS variant: identical POST /{model_id}, but settles inline and
90
+ // returns the model's direct output (no queue envelope)
91
+ // rest.fal.ai — the REST/admin surface (today: only the JWKS endpoint this twin models)
92
+ // Resolution order, per the build spec (GROUNDED from fal-js's real `withProxy` proxy protocol,
93
+ // TARGET_URL_HEADER = 'x-fal-target-url', verified against the fetched middleware.ts source):
94
+ // 1. an explicit `host` (a real Host header, or a caller-supplied routing decision)
95
+ // 2. the `x-fal-target-url` header (fal-js's OWN proxy protocol: when a caller proxies fal
96
+ // requests through their own server, the proxy middleware rewrites the request URL to the
97
+ // proxy's endpoint and stashes the ORIGINAL absolute URL in this header — a server acting as
98
+ // that proxy target recovers the true destination host from it, exactly what this twin's
99
+ // fal-sdk.integration.test.ts exercises)
100
+ // 3. no host info at all -> default 'queue', EXCEPT: `/.well-known/jwks.json` is always 'rest'
101
+ // (there is no queue/sync analog of it) and any path containing '/requests/' is always
102
+ // 'queue' (the sync surface never has sub-paths — a sync call is a bare POST /{model_id}).
103
+ export type FalSurface = 'queue' | 'sync' | 'rest';
104
+
105
+ function hostFromTargetUrlHeader(headers: Record<string, string>): string | undefined {
106
+ const raw = headers['x-fal-target-url'];
107
+ if (!raw) return undefined;
108
+ try {
109
+ return new URL(raw).host;
110
+ } catch {
111
+ return undefined;
112
+ }
113
+ }
114
+
115
+ export function routeFalSurface(req: { host?: string; headers?: Record<string, string>; path: string }): FalSurface {
116
+ const headers = lowerHeaders(req.headers);
117
+ const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
118
+ const explicitHost = req.host ?? hostFromTargetUrlHeader(headers);
119
+ if (explicitHost === 'queue.fal.run') return 'queue';
120
+ if (explicitHost === 'rest.fal.ai') return 'rest';
121
+ if (explicitHost === 'fal.run') return 'sync';
122
+ if (path === '/.well-known/jwks.json') return 'rest';
123
+ if (path.includes('/requests/')) return 'queue';
124
+ return 'queue'; // default (documented above)
125
+ }
126
+
127
+ // ── error envelopes ────────────────────────────────────────────────────────────────────────
128
+ // fal's gateway is FastAPI-based (grounded: the documented 422 validation-error shape is
129
+ // FastAPI/Pydantic's standard `{"detail": [{"loc": [...], "msg": "...", "type": "..."}]}`).
130
+ // Unmodeled/unknown-resource 404s are modeled on FastAPI's own default `{"detail": "..."}` 404
131
+ // (doc-UNVERIFIED for the EXACT message text on fal's own routes — the fact of a `{detail}` shape
132
+ // is a defensible FastAPI-framework default, not invented from nothing; capture-verify later).
133
+ function detailNotFound(message = 'Not found.'): FalResponse {
134
+ return { status: 404, body: { detail: message } };
135
+ }
136
+ function validation422(errors: Array<{ loc: string[]; msg: string; type: string }>): FalResponse {
137
+ return { status: 422, body: { detail: errors } };
138
+ }
139
+
140
+ // ── body / time / id helpers ───────────────────────────────────────────────────────────
141
+ function parseBody(body?: string): { ok: true; value: Record<string, unknown> } | { ok: false } {
142
+ if (body === undefined || body === '') return { ok: true, value: {} };
143
+ try {
144
+ const v = JSON.parse(body);
145
+ if (v && typeof v === 'object' && !Array.isArray(v)) return { ok: true, value: v as Record<string, unknown> };
146
+ return { ok: false };
147
+ } catch {
148
+ return { ok: false };
149
+ }
150
+ }
151
+ function nowIso(occurredAt?: string): string {
152
+ return occurredAt ?? new Date().toISOString();
153
+ }
154
+ function kid(type: string, id: string): string {
155
+ return `${type}:${id}`;
156
+ }
157
+
158
+ // ── projection helpers ──────────────────────────────────────────────────────────────
159
+ function rows(type: string, root?: string): Array<Record<string, unknown>> {
160
+ const prefix = `${type}:`;
161
+ return projectResources(SERVICE, root)
162
+ .filter((r) => r.type === type && r.id.startsWith(prefix) && (r as Record<string, unknown>)._deleted !== true)
163
+ .map((r) => ({ ...r, id: r.id.slice(prefix.length) }));
164
+ }
165
+ function getRow(type: string, id: string, root?: string): Record<string, unknown> | undefined {
166
+ return rows(type, root).find((r) => r.id === id);
167
+ }
168
+ function view(r: Record<string, unknown>): Record<string, unknown> {
169
+ const { type: _t, updatedAt: _u, ...rest } = r;
170
+ const out: Record<string, unknown> = {};
171
+ for (const [k, v] of Object.entries(rest)) if (!k.startsWith('_')) out[k] = v;
172
+ return out;
173
+ }
174
+
175
+ async function write(
176
+ type: string,
177
+ id: string,
178
+ fields: Record<string, unknown>,
179
+ op: string,
180
+ req: FalRequest,
181
+ ): Promise<Record<string, unknown>> {
182
+ const { resource } = await applyTwinWrite(
183
+ SERVICE,
184
+ { operation: op, subjectType: type, subjectId: kid(type, id), fields, ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' } },
185
+ req.root,
186
+ );
187
+ return view({ ...resource, id });
188
+ }
189
+
190
+ // ── GENERATIVE STUB (the twin's answer for generated output) ────────────────────────────
191
+ function canonicalJson(v: unknown): string {
192
+ if (v === null || typeof v !== 'object') return JSON.stringify(v);
193
+ if (Array.isArray(v)) return `[${v.map(canonicalJson).join(',')}]`;
194
+ const keys = Object.keys(v as Record<string, unknown>).sort();
195
+ return `{${keys.map((k) => `${JSON.stringify(k)}:${canonicalJson((v as Record<string, unknown>)[k])}`).join(',')}}`;
196
+ }
197
+ function stubOutput(modelId: string, input: unknown): string {
198
+ const hash = createHash('sha256').update(`${modelId}::${canonicalJson(input)}`).digest('hex').slice(0, 8);
199
+ return `[twin-stub:fal:${hash}] deterministic output — no GPU/model weights are run by this twin`;
200
+ }
201
+
202
+ // ── logs (grounded shape: {message, level, source:'USER', timestamp}) ────────────────────────
203
+ type FalLogLevel = 'STDERR' | 'STDOUT' | 'ERROR' | 'INFO' | 'WARN' | 'DEBUG';
204
+ function makeLog(message: string, level: FalLogLevel, occurredAt?: string): Record<string, unknown> {
205
+ return { message, level, source: 'USER', timestamp: nowIso(occurredAt) };
206
+ }
207
+
208
+ // ── request envelope views (mirror the fetched fal.ai docs' queue-status shapes) ──────────────
209
+ function requestUrls(modelId: string, requestId: string): { response_url: string; status_url: string; cancel_url: string } {
210
+ return {
211
+ response_url: `https://queue.fal.run/${modelId}/requests/${requestId}`,
212
+ status_url: `https://queue.fal.run/${modelId}/requests/${requestId}/status`,
213
+ cancel_url: `https://queue.fal.run/${modelId}/requests/${requestId}/cancel`,
214
+ };
215
+ }
216
+
217
+ function submitEnvelope(row: Record<string, unknown>, modelId: string, requestId: string): Record<string, unknown> {
218
+ return {
219
+ request_id: requestId,
220
+ // Whether the submit body itself carries `status` is doc-UNVERIFIED (the fal-js client's own
221
+ // TypeScript response types declare it; the docs' own submit example omits it) — the build
222
+ // spec's ⚠1 resolves this by preferring the client's typed shape (it is the artifact real
223
+ // callers actually depend on) and annotating the gap here rather than silently picking one.
224
+ status: row.status,
225
+ ...requestUrls(modelId, requestId),
226
+ queue_position: (row.queue_position as number) ?? 0,
227
+ };
228
+ }
229
+
230
+ function statusView(row: Record<string, unknown>, modelId: string, requestId: string, includeLogs: boolean): Record<string, unknown> {
231
+ const status = row.status as string;
232
+ const base = { status, request_id: requestId, ...requestUrls(modelId, requestId) };
233
+ if (status === 'IN_QUEUE') {
234
+ return { ...base, queue_position: (row.queue_position as number) ?? 0 };
235
+ }
236
+ // logs=0 vs unset — both currently OMIT the field (⚠8, doc-UNVERIFIED whether the vendor
237
+ // distinguishes "absent" from "explicit empty array"; capture-verify later).
238
+ const logs = includeLogs ? { logs: (row.logs as unknown[]) ?? [] } : {};
239
+ if (status === 'IN_PROGRESS') return { ...base, ...logs };
240
+ // COMPLETED
241
+ return {
242
+ ...base,
243
+ ...logs,
244
+ metrics: { inference_time: (row.inference_time as number) ?? 0.42 },
245
+ ...(row.error ? { error: row.error, error_type: row.error_type } : {}),
246
+ };
247
+ }
248
+
249
+ /**
250
+ * Deterministic, kernel-folded poll-count progression (mirrors replicate-twin.ts's pattern,
251
+ * ported to fal's 3-step lifecycle): the FIRST status poll after submit is an IN_QUEUE echo (no
252
+ * transition — real queues don't jump straight to running), the SECOND transitions to
253
+ * IN_PROGRESS with the first log line, the THIRD transitions to COMPLETED with the second log
254
+ * line + a deterministic stub output + metrics. No wall clock, no timers — purely poll-count
255
+ * folded into the kernel row, so re-running a test suite is bit-for-bit reproducible.
256
+ */
257
+ async function progressOnce(id: string, req: FalRequest): Promise<void> {
258
+ const row = getRow('request', id, req.root);
259
+ if (!row) return;
260
+ const pollCount = (row.poll_count as number) ?? 0;
261
+ if (row.status === 'IN_QUEUE') {
262
+ if (pollCount === 0) {
263
+ await write('request', id, { poll_count: pollCount + 1 }, 'request.poll', req);
264
+ return;
265
+ }
266
+ await write('request', id, {
267
+ status: 'IN_PROGRESS',
268
+ logs: [makeLog('Starting inference', 'INFO', req.occurredAt)],
269
+ poll_count: pollCount + 1,
270
+ }, 'request.progress', req);
271
+ return;
272
+ }
273
+ if (row.status === 'IN_PROGRESS') {
274
+ // The GENERATIVE CARVE-OUT is the one thing a handler may replace: the value the request
275
+ // settles to. The outcome was resolved at SUBMIT and PERSISTED (`_scripted_*`, stripped by
276
+ // `view`), so completion stays a pure fold of the log — this path never consults the engine.
277
+ const out = '_scripted_output' in row ? row._scripted_output : stubOutput(String(row.model_id ?? ''), row.input);
278
+ const priorLogs = (row.logs as unknown[]) ?? [];
279
+ await write('request', id, {
280
+ status: 'COMPLETED',
281
+ logs: [...priorLogs, makeLog('Inference complete', 'INFO', req.occurredAt)],
282
+ output: out,
283
+ completed_at: nowIso(req.occurredAt),
284
+ inference_time: typeof row._scripted_inference_time === 'number' ? row._scripted_inference_time : 0.42,
285
+ poll_count: pollCount + 1,
286
+ }, 'request.complete', req);
287
+ }
288
+ // COMPLETED: idempotent no-op — repeated polls of a settled request never regress state.
289
+ }
290
+
291
+ // ── submit (queue and sync surfaces share this; sync additionally drives it to completion) ────
292
+ async function handleSubmit(modelId: string, rawBody: string | undefined, surface: FalSurface, req: FalRequest): Promise<FalResponse> {
293
+ const parsed = parseBody(rawBody);
294
+ // The gateway's own universal validation (independent of any per-model schema, which fal's
295
+ // gateway does not itself know): the submitted input must be a JSON object. Doc-UNVERIFIED
296
+ // exact message text — the FastAPI/Pydantic `{"detail":[{loc,msg,type}]}` shape itself IS
297
+ // grounded (fal's documented 422 examples use it); this specific trigger condition is a
298
+ // plausible, defensible modeling of it, not a confirmed live-API capture.
299
+ if (!parsed.ok) {
300
+ return validation422([{ loc: ['body'], msg: 'Input should be a valid dictionary or object to extract fields from', type: 'model_attributes_type' }]);
301
+ }
302
+ const input = parsed.value;
303
+ const createdAt = nowIso(req.occurredAt);
304
+ // DETERMINISTIC (R9, resource level): the request id is a UUID-SHAPED mint over
305
+ // `${type}:${occurredAt}:${ordinal}` — the world instant the submit happened at plus the count
306
+ // of request rows already stored, read BEFORE the insert — never `randomUUID`, so two
307
+ // identical worlds mint the same id and the queue's status/result reads are byte-identical.
308
+ // Two identical submits at one instant still get distinct ids: the ordinal moved.
309
+ const ordinal = rows('request', req.root).length;
310
+ const hex = createHash('sha256').update(`request:${createdAt}:${ordinal}`).digest('hex');
311
+ const id = `${hex.slice(0, 8)}-${hex.slice(8, 12)}-4${hex.slice(13, 16)}-${((parseInt(hex[16]!, 16) & 0x3) | 0x8).toString(16)}${hex.slice(17, 20)}-${hex.slice(20, 32)}`;
312
+ // Scenario: resolve any scripted outcome NOW (submit time) and persist it on the resource, so
313
+ // the status progression is a pure fold of the log. A miss leaves the labeled stub in place.
314
+ const scripted: FalScenarioRespond | null = req.scenarioEngine
315
+ ? (() => {
316
+ const decision = req.scenarioEngine!.next({ modelId, input, surface: surface === 'sync' ? 'sync' : 'queue', nthRequest: ordinal + 1 });
317
+ return decision.kind === 'handler' ? (decision.respond as FalScenarioRespond) : null;
318
+ })()
319
+ : null;
320
+ await write('request', id, {
321
+ model_id: modelId,
322
+ input,
323
+ ...(scripted ? { _scripted_output: scripted.output } : {}),
324
+ ...(scripted?.inferenceTime !== undefined ? { _scripted_inference_time: scripted.inferenceTime } : {}),
325
+ status: 'IN_QUEUE',
326
+ queue_position: 0,
327
+ logs: [],
328
+ output: null,
329
+ error: null,
330
+ error_type: null,
331
+ created_at: createdAt,
332
+ completed_at: null,
333
+ webhook_url: null,
334
+ poll_count: 0,
335
+ }, 'request.create', req);
336
+
337
+ if (surface === 'sync') {
338
+ // fal.run settles INLINE (no queue polling exposed) and returns the model's direct output —
339
+ // no queue envelope. Drive the SAME deterministic 3-step progression the queue surface takes
340
+ // across 3 polls (poll1 IN_QUEUE echo -> poll2 IN_PROGRESS -> poll3 COMPLETED) in one call —
341
+ // mirrors replicate's `Prefer: wait` inline-settle pattern.
342
+ await progressOnce(id, req);
343
+ await progressOnce(id, req);
344
+ await progressOnce(id, req);
345
+ const settled = getRow('request', id, req.root)!;
346
+ return { status: 200, body: { output: settled.output } };
347
+ }
348
+ const row = getRow('request', id, req.root)!;
349
+ return { status: 200, body: submitEnvelope(row, modelId, id) };
350
+ }
351
+
352
+ async function handleStatus(modelId: string, requestId: string, includeLogs: boolean, req: FalRequest): Promise<FalResponse> {
353
+ const existing = getRow('request', requestId, req.root);
354
+ if (!existing) return detailNotFound('Request not found.');
355
+ await progressOnce(requestId, req);
356
+ const row = getRow('request', requestId, req.root)!;
357
+ return { status: 200, body: statusView(row, modelId, requestId, includeLogs) };
358
+ }
359
+
360
+ async function handleResult(requestId: string, req: FalRequest): Promise<FalResponse> {
361
+ const row = getRow('request', requestId, req.root);
362
+ if (!row) return detailNotFound('Request not found.');
363
+ // Pre-completion behavior is genuinely unmodeled (`fal.queue.response_long_poll`, todo — this
364
+ // twin settles instantly on `status` polls rather than long-polling); every DONE capability
365
+ // that reads this route first drives the request to COMPLETED via `status` polls.
366
+ return { status: 200, body: { output: row.output ?? null } };
367
+ }
368
+
369
+ async function handleCancel(requestId: string, req: FalRequest): Promise<FalResponse> {
370
+ const row = getRow('request', requestId, req.root);
371
+ // Cancel's 404 is its OWN documented shape (`{"status":"NOT_FOUND"}`), NOT the generic
372
+ // `{detail}` 404 envelope every other unknown-resource route in this pack uses — grounded
373
+ // verbatim from fal.ai/docs/model-endpoints/queue's cancel-endpoint error table.
374
+ if (!row) return { status: 404, body: { status: 'NOT_FOUND' } };
375
+ if (row.status === 'COMPLETED') return { status: 400, body: { status: 'ALREADY_COMPLETED' } };
376
+ // The real terminal state a successful cancellation settles into is UNKNOWN (fal's documented
377
+ // status enum — IN_QUEUE/IN_PROGRESS/COMPLETED — has no CANCELLED member; `⚠4` in the build
378
+ // spec). This twin claims ONLY the grounded 202 acknowledgement body, not any resulting state
379
+ // transition (`fal.queue.cancelled_terminal_state` stays `todo`) — deliberately does NOT
380
+ // mutate `row.status`, so a subsequent status poll continues its ordinary progression rather
381
+ // than asserting an invented terminal state.
382
+ return { status: 202, body: { status: 'CANCELLATION_REQUESTED' } };
383
+ }
384
+
385
+ // ── route handler ────────────────────────────────────────────────────────────────────
386
+ export async function handleFalTwinRequest(req: FalRequest): Promise<FalResponse> {
387
+ const method = req.method.toUpperCase();
388
+ const [rawPath, rawQuery] = req.path.split('?');
389
+ const path = (rawPath ?? '/').replace(/\/+$/, '') || '/';
390
+ const headers = lowerHeaders(req.headers);
391
+ const surface = routeFalSurface({ host: req.host, headers: req.headers, path });
392
+
393
+ if (req.readOnly && method !== 'GET' && method !== 'HEAD') {
394
+ return { status: 405, body: { detail: 'twin is read-only; omit readOnly to accept writes' } };
395
+ }
396
+
397
+ if (path === '/' || path === '') {
398
+ return { status: 200, body: { service: 'fal', object: 'twin' } };
399
+ }
400
+
401
+ // ── REST surface: today, only the JWKS endpoint ──
402
+ if (surface === 'rest') {
403
+ if (method === 'GET' && path === '/.well-known/jwks.json') {
404
+ return { status: 200, body: await falJwks(req.root) };
405
+ }
406
+ return detailNotFound();
407
+ }
408
+
409
+ const seg = path.replace(/^\/+/, '').split('/').filter(Boolean);
410
+ const reqIdx = seg.indexOf('requests');
411
+
412
+ // ── submit: POST /{model_id} (model ids are 2- or 3-segment, e.g. fal-ai/flux/schnell) ──
413
+ if (reqIdx === -1) {
414
+ if (method === 'POST' && seg.length >= 1) {
415
+ return handleSubmit(seg.map(decodeURIComponent).join('/'), req.body, surface, req);
416
+ }
417
+ return detailNotFound();
418
+ }
419
+
420
+ // ── queue lifecycle: {model_id}/requests/{request_id}[/status[/stream]|/response|/cancel] ──
421
+ const modelId = seg.slice(0, reqIdx).map(decodeURIComponent).join('/');
422
+ const requestId = decodeURIComponent(seg[reqIdx + 1] ?? '');
423
+ const rest = seg.slice(reqIdx + 2);
424
+
425
+ if (method === 'GET' && rest.length === 1 && rest[0] === 'status') {
426
+ const query = new URLSearchParams(rawQuery ?? '');
427
+ const includeLogs = query.get('logs') === '1';
428
+ return handleStatus(modelId, requestId, includeLogs, req);
429
+ }
430
+ if (method === 'GET' && rest.length === 2 && rest[0] === 'status' && rest[1] === 'stream') {
431
+ // SSE streaming (`fal.streaming.status_stream_sse`, todo) — genuinely unmodeled.
432
+ return detailNotFound();
433
+ }
434
+ if (method === 'GET' && rest.length === 0) {
435
+ return handleResult(requestId, req);
436
+ }
437
+ if (method === 'GET' && rest.length === 1 && rest[0] === 'response') {
438
+ // The doc-example alias route — served so a doc-literal caller works too (⚠2).
439
+ return handleResult(requestId, req);
440
+ }
441
+ if (method === 'PUT' && rest.length === 1 && rest[0] === 'cancel') {
442
+ return handleCancel(requestId, req);
443
+ }
444
+
445
+ return detailNotFound();
446
+ }
447
+
448
+ export type FalTwinSnapshot = {
449
+ resourceTypes: readonly FalResourceType[];
450
+ implementedEndpoints: readonly string[];
451
+ };
452
+
453
+ // Endpoint inventory used by the conformance snapshot (self-referential — see fal-conformance.ts
454
+ // header note and docs/contributing/conformance.md's "2 spec" discussion of this pattern, shared with replicate/
455
+ // elevenlabs/polar). fal's real v1 surface is intentionally much smaller than replicate's (no
456
+ // models/versions/collections/trainings/deployments CRUD — build spec §0), so this snapshot
457
+ // counts each INDIVIDUALLY-DOCUMENTED request/response CONTRACT (not just each distinct URL
458
+ // pattern) as its own entry — e.g. cancel's three distinct, separately-grounded status-code
459
+ // contracts (202/400/404) are 3 entries, not 1 — since each is independently verified by its own
460
+ // `done` capability and lumping them would UNDER-count real, grounded, implemented behavior
461
+ // rather than pad it.
462
+ export function falTwinSnapshot(): FalTwinSnapshot {
463
+ return {
464
+ resourceTypes: FAL_RESOURCE_TYPES,
465
+ implementedEndpoints: [
466
+ 'POST queue.fal.run/{model_id} (queue submit)',
467
+ 'POST fal.run/{model_id} (sync submit, direct output)',
468
+ 'GET queue.fal.run/{model_id}/requests/{request_id}/status (IN_QUEUE/IN_PROGRESS/COMPLETED progression, logs param)',
469
+ 'GET queue.fal.run/{model_id}/requests/{request_id}/status (404 unknown request_id)',
470
+ 'GET queue.fal.run/{model_id}/requests/{request_id} (result, client-built bare form)',
471
+ 'GET queue.fal.run/{model_id}/requests/{request_id}/response (result, docs-example alias form)',
472
+ 'PUT queue.fal.run/{model_id}/requests/{request_id}/cancel (202 CANCELLATION_REQUESTED)',
473
+ 'PUT queue.fal.run/{model_id}/requests/{request_id}/cancel (400 ALREADY_COMPLETED)',
474
+ 'PUT queue.fal.run/{model_id}/requests/{request_id}/cancel (404 NOT_FOUND)',
475
+ 'POST queue.fal.run/{model_id} (422 validation error)',
476
+ 'GET rest.fal.ai/.well-known/jwks.json',
477
+ ],
478
+ };
479
+ }