@volter/twin-fal 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,26 @@
1
+ import { type PackScenarioAdapter, type ScenarioDocument, ScenarioEngine } from '@volter/world-core';
2
+ /** The request slice the scenario system sees — built by the submit route from the parsed body. */
3
+ export type FalScenarioRequest = {
4
+ modelId: string;
5
+ input: Record<string, unknown>;
6
+ /** The surface the submit arrived on: fal's queue (queue.fal.run) or the inline fal.run. */
7
+ surface: 'queue' | 'sync';
8
+ /** 1-based ordinal of this submit among the request rows in the root. */
9
+ nthRequest: number;
10
+ };
11
+ export type FalScenarioEngine = ScenarioEngine<FalScenarioRequest>;
12
+ /** The scripted outcome: the `output` the request settles to (any JSON value the model would
13
+ * return — real fal models answer objects like `{ images: [...] }`, not only strings), and
14
+ * optionally the reported inference time. */
15
+ export type FalScenarioRespond = {
16
+ output: unknown;
17
+ inferenceTime?: number;
18
+ };
19
+ /** Stable text of a submitted input — the corpus `inputIncludes` and `textPattern` extractors
20
+ * read. Key order is sorted so the same input always yields the same corpus. */
21
+ export declare function canonicalInputText(input: unknown): string;
22
+ export declare const falScenarioAdapter: PackScenarioAdapter<FalScenarioRequest>;
23
+ /** Load + strictly validate a handlers document (handlers/fal.json). A broken file fails server
24
+ * construction loudly; it never falls back or misfires silently. */
25
+ export declare function loadFalScenarioDocument(path: string): ScenarioDocument;
26
+ export declare function createFalScenarioEngine(document?: ScenarioDocument): FalScenarioEngine;
@@ -0,0 +1,100 @@
1
+ // The fal pack's HALF of the scenario system, on the kernel's ONE engine (@volter/world-core
2
+ // scenario.ts). The GRAMMAR (ordering, once/scope/phase, extractors, strict parsing, miss
3
+ // records, twin.use) is the kernel's and identical for every vendor; this module declares only
4
+ // the VOCABULARY:
5
+ // • the `on` keys a handler may match (model id, submitted input text, the queue-vs-sync
6
+ // surface, and the 1-based ordinal of the submit within the world),
7
+ // • what a `respond` payload may contain ({ output, inferenceTime? }),
8
+ // • nothing else — fal's PROTOCOL envelope (the IN_QUEUE → IN_PROGRESS → COMPLETED machine,
9
+ // queue_position, logs, urls, metrics) stays the twin's, faithful and unscriptable. What a
10
+ // handler replaces is exactly the GENERATIVE STUB: the `output` a request settles to,
11
+ // which without a handler is the labeled `[twin-stub:fal:<hash8>]` value.
12
+ //
13
+ // Matching happens at SUBMIT (POST /{model_id}, both the queue and the fal.run sync surface) and
14
+ // the outcome is PERSISTED on the request resource, so completion stays a pure fold of the kernel
15
+ // log — the status progression never consults the engine and holds no hidden session state.
16
+ // `nthRequest` is a PACK matcher over the count of request rows already in the root, not the
17
+ // engine's call counter.
18
+ //
19
+ // The handler FILE (handlers/fal.json in a world dir) is the only write surface.
20
+ import { getActiveWorldStore, parseScenarioDocument, ScenarioError, ScenarioEngine } from '@volter/world-core';
21
+ const RESPOND_KEYS = new Set(['output', 'inferenceTime']);
22
+ const nonEmptyString = (cond) => typeof cond === 'string' && cond.length > 0;
23
+ /** Stable text of a submitted input — the corpus `inputIncludes` and `textPattern` extractors
24
+ * read. Key order is sorted so the same input always yields the same corpus. */
25
+ export function canonicalInputText(input) {
26
+ if (input === null || typeof input !== 'object')
27
+ return JSON.stringify(input) ?? 'null';
28
+ if (Array.isArray(input))
29
+ return `[${input.map(canonicalInputText).join(',')}]`;
30
+ const keys = Object.keys(input).sort();
31
+ return `{${keys.map((k) => `${JSON.stringify(k)}:${canonicalInputText(input[k])}`).join(',')}}`;
32
+ }
33
+ export const falScenarioAdapter = {
34
+ vendor: 'fal',
35
+ features: (req) => ({
36
+ modelId: req.modelId,
37
+ input: canonicalInputText(req.input).slice(0, 300),
38
+ surface: req.surface,
39
+ nthRequest: req.nthRequest,
40
+ }),
41
+ matchers: {
42
+ modelIdEquals: (req, cond) => nonEmptyString(cond) && req.modelId === cond,
43
+ modelIdIncludes: (req, cond) => nonEmptyString(cond) && req.modelId.toLowerCase().includes(cond.toLowerCase()),
44
+ inputIncludes: (req, cond) => nonEmptyString(cond) && canonicalInputText(req.input).toLowerCase().includes(cond.toLowerCase()),
45
+ surfaceEquals: (req, cond) => (cond === 'queue' || cond === 'sync') && req.surface === cond,
46
+ // The 1-based index among requests submitted into the ROOT, not the engine's call counter —
47
+ // no hidden session state.
48
+ nthRequest: (req, cond) => Number.isInteger(cond) && req.nthRequest === cond,
49
+ },
50
+ text: (req) => canonicalInputText(req.input),
51
+ validateOn: (on) => {
52
+ for (const k of ['modelIdEquals', 'modelIdIncludes', 'inputIncludes']) {
53
+ if (on[k] !== undefined && (typeof on[k] !== 'string' || !on[k]))
54
+ return `on.${k} is a non-empty string`;
55
+ }
56
+ if (on.surfaceEquals !== undefined && on.surfaceEquals !== 'queue' && on.surfaceEquals !== 'sync')
57
+ return 'on.surfaceEquals is "queue" or "sync"';
58
+ if (on.nthRequest !== undefined && (!Number.isInteger(on.nthRequest) || on.nthRequest < 1))
59
+ return 'on.nthRequest is a positive integer';
60
+ return null;
61
+ },
62
+ validateRespond: (respond) => {
63
+ if (typeof respond !== 'object' || respond === null || Array.isArray(respond))
64
+ return 'respond is an object { output, inferenceTime? }';
65
+ const r = respond;
66
+ for (const k of Object.keys(r))
67
+ if (!RESPOND_KEYS.has(k))
68
+ return `respond: unknown key "${k}" (valid: ${[...RESPOND_KEYS].join(', ')})`;
69
+ if (!('output' in r))
70
+ return 'respond needs output (the value the request settles to — any JSON the model would return)';
71
+ if (r.output === undefined)
72
+ return 'respond.output is a JSON value, not undefined';
73
+ if (r.inferenceTime !== undefined && (typeof r.inferenceTime !== 'number' || !Number.isFinite(r.inferenceTime) || r.inferenceTime < 0)) {
74
+ return 'respond.inferenceTime is a non-negative number of SECONDS';
75
+ }
76
+ return null;
77
+ },
78
+ };
79
+ /** Load + strictly validate a handlers document (handlers/fal.json). A broken file fails server
80
+ * construction loudly; it never falls back or misfires silently. */
81
+ export function loadFalScenarioDocument(path) {
82
+ let parsed;
83
+ try {
84
+ // Read through the ACTIVE WorldStore, never the filesystem directly (runtime contract R12b):
85
+ // the handlers document is WORLD STATE, so a MemoryWorldStore / DO-backed world serves ITS
86
+ // OWN scenario instead of whatever happens to sit on the host disk — and the serve path
87
+ // stays workerd-clean.
88
+ const raw = getActiveWorldStore().read(path);
89
+ if (raw === null)
90
+ throw new Error(`ENOENT: no such file or directory, open '${path}'`);
91
+ parsed = JSON.parse(raw);
92
+ }
93
+ catch (e) {
94
+ throw new ScenarioError(`fal scenario: cannot read/parse ${path}: ${e instanceof Error ? e.message : String(e)}`);
95
+ }
96
+ return parseScenarioDocument(parsed, falScenarioAdapter);
97
+ }
98
+ export function createFalScenarioEngine(document) {
99
+ return new ScenarioEngine(falScenarioAdapter, document);
100
+ }
@@ -0,0 +1,18 @@
1
+ /** Options every fal-twin HTTP surface needs, independent of who owns the socket. */
2
+ export interface FalTwinFetchOptions {
3
+ root?: string;
4
+ readOnly?: boolean;
5
+ /** Path to the world dir's handlers/fal.json — read ONCE, at construction, through the active
6
+ * WorldStore. Absent → no engine, and every request settles to the labeled stub. */
7
+ scenarioPath?: string;
8
+ }
9
+ export declare function createFalTwinFetch(options?: FalTwinFetchOptions): (request: Request) => Promise<Response>;
10
+ export declare function createFalTwinServer(options?: {
11
+ root?: string;
12
+ port?: number;
13
+ readOnly?: boolean;
14
+ scenarioPath?: string;
15
+ }): Promise<{
16
+ port: number;
17
+ stop: () => void;
18
+ }>;
@@ -0,0 +1,53 @@
1
+ // fal twin HTTP server — serve the full fal API twin handler over HTTP so the real
2
+ // `@fal-ai/client` SDK (via a `requestMiddleware` that rewrites the URL to this server — see
3
+ // fal-sdk.integration.test.ts) works unmodified against it. Every route is plain JSON. Request
4
+ // headers are forwarded verbatim — NOTE the server deliberately does NOT thread the incoming
5
+ // HTTP `Host` header as `handleFalTwinRequest`'s `host` param: once a caller's request has been
6
+ // rewritten to point at THIS server (127.0.0.1:<port>), the literal Host header is always
7
+ // 127.0.0.1, which carries no vendor-host information. Host-split routing instead reads the
8
+ // `x-fal-target-url` header (the real fal-js proxy protocol, which carries the ORIGINAL absolute
9
+ // URL) — see fal-twin.ts's `routeFalSurface`. Writable by default; pass readOnly to reject writes
10
+ // with 405 (D3). State is the kernel projection (no side-store) — see fal-twin.ts.
11
+ //
12
+ // FETCH-FIRST (runtime contract R12b): the serve path is the plain fetch below, built from the
13
+ // kernel's ONE adaptation (`createTwinFetchFromHandler` — the keyless read doors, body read,
14
+ // header map, worldNow() stamp, JSON reply); this file contributes only VALUES. The server is one
15
+ // line of Bun.serve around that same closure.
16
+ //
17
+ // THE READ DOORS (runtime contract R5): keyless `GET /twin` (the manifest, which teaches the
18
+ // handler shape) and `GET /twin/scenario` (active handlers + match/miss counts). Both are the
19
+ // adapter's, answered before any vendor routing, and neither takes a credential.
20
+ //
21
+ // SCENARIO scripting (fal-scenario.ts): pass `scenarioPath` (or set the TWIN_FAL_SCENARIO env
22
+ // var) to script what a submitted request settles to. A malformed scenario throws at
23
+ // construction — loudly, never a silent fallback.
24
+ import { serveHttp } from '@volter/world-core';
25
+ import { handleFalTwinRequest } from "./fal-twin.js";
26
+ import { createTwinFetchFromHandler, twinManifest } from '@volter/world-core';
27
+ import { createFalScenarioEngine, loadFalScenarioDocument } from "./fal-scenario.js";
28
+ export function createFalTwinFetch(options = {}) {
29
+ const scenarioPath = options.scenarioPath ?? process.env.TWIN_FAL_SCENARIO;
30
+ const scenarioEngine = scenarioPath ? createFalScenarioEngine(loadFalScenarioDocument(scenarioPath)) : undefined;
31
+ const { scenarioPath: _scenarioPath, ...handlerOptions } = options;
32
+ return createTwinFetchFromHandler(handleFalTwinRequest, {
33
+ ...handlerOptions,
34
+ manifest: () => twinManifest({
35
+ vendor: 'fal',
36
+ twinOf: 'the fal.ai inference queue',
37
+ stateSentence: 'Stateful: it stores queued inference requests and their status/log/response lifecycle. Create them through the vendor\'s OWN API (POST /{model_id} on queue.fal.run or fal.run) with any credential.',
38
+ behaviorSentence: 'The PROTOCOL envelope (IN_QUEUE → IN_PROGRESS → COMPLETED, queue_position, logs, urls, metrics) is the twin\'s and faithful. What a request SETTLES TO is scripted by MSW-shaped handlers in the world dir (handlers/fal.json): {on:{modelIdEquals|modelIdIncludes|inputIncludes|surfaceEquals|nthRequest|nthCall}, respond:{output, inferenceTime?}, once?, phase?}. Matching happens at submit and the outcome persists on the request; unmatched submits settle to the labeled [twin-stub:fal:…] value, and the misses are inspectable at GET /twin/scenario.',
39
+ exampleHandler: { on: { modelIdIncludes: 'flux' }, respond: { output: { images: [{ url: 'https://cdn.example.com/scripted.png' }] } } },
40
+ engine: scenarioEngine,
41
+ }),
42
+ scenarioStatus: () => (scenarioEngine ? scenarioEngine.status() : { vendor: 'fal', handlers: [], misses: 0, recentMisses: [] }),
43
+ ...(scenarioEngine ? { handlerOptions: { scenarioEngine } } : {}),
44
+ });
45
+ }
46
+ export async function createFalTwinServer(options = {}) {
47
+ const server = await serveHttp({
48
+ port: options.port ?? 0,
49
+ idleTimeout: 60,
50
+ fetch: createFalTwinFetch(options),
51
+ });
52
+ return { port: server.port ?? options.port ?? 0, stop: () => server.stop(true) };
53
+ }
@@ -0,0 +1,36 @@
1
+ import type { FalScenarioEngine } from './fal-scenario.js';
2
+ export type FalRequest = {
3
+ method: string;
4
+ path: string;
5
+ body?: string;
6
+ /** Explicit host (e.g. from an HTTP `Host` header or a direct routing decision). Takes
7
+ * priority over `x-fal-target-url` in `headers`. */
8
+ host?: string;
9
+ headers?: Record<string, string>;
10
+ occurredAt?: string;
11
+ root?: string;
12
+ readOnly?: boolean;
13
+ /** The pack's scenario engine, mounted by the fetch adapter when a world dir carries a
14
+ * handlers/fal.json. Absent → every request settles to the labeled deterministic stub,
15
+ * unchanged. Consulted at SUBMIT only; the outcome persists on the request resource. */
16
+ scenarioEngine?: FalScenarioEngine;
17
+ };
18
+ export type FalResponse = {
19
+ status: number;
20
+ body: unknown;
21
+ headers?: Record<string, string>;
22
+ };
23
+ export declare const FAL_RESOURCE_TYPES: readonly ["request", "signing_key"];
24
+ export type FalResourceType = typeof FAL_RESOURCE_TYPES[number];
25
+ export type FalSurface = 'queue' | 'sync' | 'rest';
26
+ export declare function routeFalSurface(req: {
27
+ host?: string;
28
+ headers?: Record<string, string>;
29
+ path: string;
30
+ }): FalSurface;
31
+ export declare function handleFalTwinRequest(req: FalRequest): Promise<FalResponse>;
32
+ export type FalTwinSnapshot = {
33
+ resourceTypes: readonly FalResourceType[];
34
+ implementedEndpoints: readonly string[];
35
+ };
36
+ export declare function falTwinSnapshot(): FalTwinSnapshot;
@@ -0,0 +1,414 @@
1
+ // fal.ai API twin REQUEST HANDLER — a v1 slice of the fal.ai queue-lifecycle model-inference
2
+ // surface, backed by the event/action-log kernel (@volter/world-core). Contract:
3
+ // handleFalTwinRequest({ method, path, body, host?, headers, root, readOnly }) -> { status, body, headers }
4
+ //
5
+ // SOURCE OF TRUTH: grounded read-only against fal.ai's own published docs (fal.ai/docs/model-
6
+ // endpoints/queue, fal.ai/docs/model-endpoints/webhooks) and the fal-js client source
7
+ // (github.com/fal-ai/fal-js — libs/client/src/{config,middleware,queue,request}.ts), fetched
8
+ // during this build (see spec-sources.json). Every route/field/status below cites one of those
9
+ // two sources; a doc-UNVERIFIED shape is annotated inline (⚠) and NOT claimed beyond what the
10
+ // verify() actually proves.
11
+ //
12
+ // Facts grounded from fal.ai/docs/model-endpoints/queue:
13
+ // - submit: POST https://queue.fal.run/{model_id} -> {request_id, response_url, status_url,
14
+ // cancel_url, queue_position}. status: GET .../requests/{request_id}/status[?logs=1] ->
15
+ // {status: IN_QUEUE|IN_PROGRESS|COMPLETED, request_id, response_url, queue_position
16
+ // (IN_QUEUE only), logs (IN_PROGRESS/COMPLETED), metrics.inference_time (COMPLETED only)}.
17
+ // - cancel: PUT .../requests/{request_id}/cancel -> 202 {"status":"CANCELLATION_REQUESTED"} |
18
+ // 400 {"status":"ALREADY_COMPLETED"} | 404 {"status":"NOT_FOUND"} (NOT a {detail} envelope —
19
+ // this exact shape is doc-grounded, distinct from every other 404 in this pack).
20
+ // - result: docs literally show GET .../requests/{request_id}/response, but the REAL fal-js
21
+ // client builds the result URL WITHOUT the /response suffix (verified against the fetched
22
+ // queue.ts source: `path: /requests/${requestId}`) — a genuine doc/client conflict (⚠2 in the
23
+ // build spec). This twin serves BOTH forms so either a doc-literal caller or the real SDK
24
+ // works; the /response form is the doc's own example, still exercised, but not separately
25
+ // claimed as its own capability (`fal.queue.response_alias_route` stays `todo`).
26
+ // Facts grounded from fal.ai/docs/model-endpoints/webhooks — see fal-webhooks.ts.
27
+ // Facts grounded from fal-js middleware.ts/config.ts/request.ts — see fal-twin.ts's
28
+ // `routeFalSurface` and fal-sdk.integration.test.ts.
29
+ //
30
+ // State lives ENTIRELY in the kernel action log: writes go through `applyTwinWrite`, reads are
31
+ // the projection (`projectResources`). There is NO Map/array side-store (D1). No real fal is
32
+ // ever contacted.
33
+ //
34
+ // THE GENERATIVE STUB (D2 honesty): fal is a GENERATIVE vendor — model inference requires
35
+ // hosted GPU weights that cannot run locally, so a request's `output` is a clearly-labeled
36
+ // DETERMINISTIC stub (`[twin-stub:fal:<hash8>]`), derived from hash(model_id, canonical-
37
+ // JSON(input)) — same input twice -> identical output, different input -> different output. The
38
+ // PROTOCOL ENVELOPE (status machine, queue_position, logs, urls, metrics shape) is faithful; only
39
+ // the generated content is a stub.
40
+ //
41
+ // Honesty (D2): an unmodeled route returns a FastAPI-style `{"detail": "..."}` 404 (fal's
42
+ // gateway is FastAPI-based per its documented 422 validation-error shape), never a fabricated
43
+ // success. readOnly rejects writes with 405.
44
+ //
45
+ // Kernel SUBJECT ids are type-prefixed (`request:<uuid>`) — the PUBLIC request_id emitted to
46
+ // clients is the bare uuid. `kid()` builds the subject id; `rows()` strips the prefix back off.
47
+ import { createHash } from 'node:crypto';
48
+ import { applyTwinWrite, projectResources } from '@volter/world-core';
49
+ import { falJwks } from "./fal-webhooks.js";
50
+ const SERVICE = 'fal';
51
+ // Resource types the twin actually PROJECTS in the kernel — the honest v1 subset. `signing_key`
52
+ // is a lazily-materialized per-root singleton (the same "fold on first touch" pattern replicate's
53
+ // webhook_secret / elevenlabs' dubbing.get use); it's never itself a REST-addressable resource
54
+ // (fal has no admin surface to list/rotate keys — see README ## Coverage), only the ed25519
55
+ // keypair backing the JWKS endpoint + webhook signing.
56
+ export const FAL_RESOURCE_TYPES = ['request', 'signing_key'];
57
+ function lowerHeaders(h) {
58
+ const out = {};
59
+ for (const [k, v] of Object.entries(h ?? {}))
60
+ out[k.toLowerCase()] = v;
61
+ return out;
62
+ }
63
+ function hostFromTargetUrlHeader(headers) {
64
+ const raw = headers['x-fal-target-url'];
65
+ if (!raw)
66
+ return undefined;
67
+ try {
68
+ return new URL(raw).host;
69
+ }
70
+ catch {
71
+ return undefined;
72
+ }
73
+ }
74
+ export function routeFalSurface(req) {
75
+ const headers = lowerHeaders(req.headers);
76
+ const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
77
+ const explicitHost = req.host ?? hostFromTargetUrlHeader(headers);
78
+ if (explicitHost === 'queue.fal.run')
79
+ return 'queue';
80
+ if (explicitHost === 'rest.fal.ai')
81
+ return 'rest';
82
+ if (explicitHost === 'fal.run')
83
+ return 'sync';
84
+ if (path === '/.well-known/jwks.json')
85
+ return 'rest';
86
+ if (path.includes('/requests/'))
87
+ return 'queue';
88
+ return 'queue'; // default (documented above)
89
+ }
90
+ // ── error envelopes ────────────────────────────────────────────────────────────────────────
91
+ // fal's gateway is FastAPI-based (grounded: the documented 422 validation-error shape is
92
+ // FastAPI/Pydantic's standard `{"detail": [{"loc": [...], "msg": "...", "type": "..."}]}`).
93
+ // Unmodeled/unknown-resource 404s are modeled on FastAPI's own default `{"detail": "..."}` 404
94
+ // (doc-UNVERIFIED for the EXACT message text on fal's own routes — the fact of a `{detail}` shape
95
+ // is a defensible FastAPI-framework default, not invented from nothing; capture-verify later).
96
+ function detailNotFound(message = 'Not found.') {
97
+ return { status: 404, body: { detail: message } };
98
+ }
99
+ function validation422(errors) {
100
+ return { status: 422, body: { detail: errors } };
101
+ }
102
+ // ── body / time / id helpers ───────────────────────────────────────────────────────────
103
+ function parseBody(body) {
104
+ if (body === undefined || body === '')
105
+ return { ok: true, value: {} };
106
+ try {
107
+ const v = JSON.parse(body);
108
+ if (v && typeof v === 'object' && !Array.isArray(v))
109
+ return { ok: true, value: v };
110
+ return { ok: false };
111
+ }
112
+ catch {
113
+ return { ok: false };
114
+ }
115
+ }
116
+ function nowIso(occurredAt) {
117
+ return occurredAt ?? new Date().toISOString();
118
+ }
119
+ function kid(type, id) {
120
+ return `${type}:${id}`;
121
+ }
122
+ // ── projection helpers ──────────────────────────────────────────────────────────────
123
+ function rows(type, root) {
124
+ const prefix = `${type}:`;
125
+ return projectResources(SERVICE, root)
126
+ .filter((r) => r.type === type && r.id.startsWith(prefix) && r._deleted !== true)
127
+ .map((r) => ({ ...r, id: r.id.slice(prefix.length) }));
128
+ }
129
+ function getRow(type, id, root) {
130
+ return rows(type, root).find((r) => r.id === id);
131
+ }
132
+ function view(r) {
133
+ const { type: _t, updatedAt: _u, ...rest } = r;
134
+ const out = {};
135
+ for (const [k, v] of Object.entries(rest))
136
+ if (!k.startsWith('_'))
137
+ out[k] = v;
138
+ return out;
139
+ }
140
+ async function write(type, id, fields, op, req) {
141
+ const { resource } = await applyTwinWrite(SERVICE, { operation: op, subjectType: type, subjectId: kid(type, id), fields, ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' } }, req.root);
142
+ return view({ ...resource, id });
143
+ }
144
+ // ── GENERATIVE STUB (the twin's answer for generated output) ────────────────────────────
145
+ function canonicalJson(v) {
146
+ if (v === null || typeof v !== 'object')
147
+ return JSON.stringify(v);
148
+ if (Array.isArray(v))
149
+ return `[${v.map(canonicalJson).join(',')}]`;
150
+ const keys = Object.keys(v).sort();
151
+ return `{${keys.map((k) => `${JSON.stringify(k)}:${canonicalJson(v[k])}`).join(',')}}`;
152
+ }
153
+ function stubOutput(modelId, input) {
154
+ const hash = createHash('sha256').update(`${modelId}::${canonicalJson(input)}`).digest('hex').slice(0, 8);
155
+ return `[twin-stub:fal:${hash}] deterministic output — no GPU/model weights are run by this twin`;
156
+ }
157
+ function makeLog(message, level, occurredAt) {
158
+ return { message, level, source: 'USER', timestamp: nowIso(occurredAt) };
159
+ }
160
+ // ── request envelope views (mirror the fetched fal.ai docs' queue-status shapes) ──────────────
161
+ function requestUrls(modelId, requestId) {
162
+ return {
163
+ response_url: `https://queue.fal.run/${modelId}/requests/${requestId}`,
164
+ status_url: `https://queue.fal.run/${modelId}/requests/${requestId}/status`,
165
+ cancel_url: `https://queue.fal.run/${modelId}/requests/${requestId}/cancel`,
166
+ };
167
+ }
168
+ function submitEnvelope(row, modelId, requestId) {
169
+ return {
170
+ request_id: requestId,
171
+ // Whether the submit body itself carries `status` is doc-UNVERIFIED (the fal-js client's own
172
+ // TypeScript response types declare it; the docs' own submit example omits it) — the build
173
+ // spec's ⚠1 resolves this by preferring the client's typed shape (it is the artifact real
174
+ // callers actually depend on) and annotating the gap here rather than silently picking one.
175
+ status: row.status,
176
+ ...requestUrls(modelId, requestId),
177
+ queue_position: row.queue_position ?? 0,
178
+ };
179
+ }
180
+ function statusView(row, modelId, requestId, includeLogs) {
181
+ const status = row.status;
182
+ const base = { status, request_id: requestId, ...requestUrls(modelId, requestId) };
183
+ if (status === 'IN_QUEUE') {
184
+ return { ...base, queue_position: row.queue_position ?? 0 };
185
+ }
186
+ // logs=0 vs unset — both currently OMIT the field (⚠8, doc-UNVERIFIED whether the vendor
187
+ // distinguishes "absent" from "explicit empty array"; capture-verify later).
188
+ const logs = includeLogs ? { logs: row.logs ?? [] } : {};
189
+ if (status === 'IN_PROGRESS')
190
+ return { ...base, ...logs };
191
+ // COMPLETED
192
+ return {
193
+ ...base,
194
+ ...logs,
195
+ metrics: { inference_time: row.inference_time ?? 0.42 },
196
+ ...(row.error ? { error: row.error, error_type: row.error_type } : {}),
197
+ };
198
+ }
199
+ /**
200
+ * Deterministic, kernel-folded poll-count progression (mirrors replicate-twin.ts's pattern,
201
+ * ported to fal's 3-step lifecycle): the FIRST status poll after submit is an IN_QUEUE echo (no
202
+ * transition — real queues don't jump straight to running), the SECOND transitions to
203
+ * IN_PROGRESS with the first log line, the THIRD transitions to COMPLETED with the second log
204
+ * line + a deterministic stub output + metrics. No wall clock, no timers — purely poll-count
205
+ * folded into the kernel row, so re-running a test suite is bit-for-bit reproducible.
206
+ */
207
+ async function progressOnce(id, req) {
208
+ const row = getRow('request', id, req.root);
209
+ if (!row)
210
+ return;
211
+ const pollCount = row.poll_count ?? 0;
212
+ if (row.status === 'IN_QUEUE') {
213
+ if (pollCount === 0) {
214
+ await write('request', id, { poll_count: pollCount + 1 }, 'request.poll', req);
215
+ return;
216
+ }
217
+ await write('request', id, {
218
+ status: 'IN_PROGRESS',
219
+ logs: [makeLog('Starting inference', 'INFO', req.occurredAt)],
220
+ poll_count: pollCount + 1,
221
+ }, 'request.progress', req);
222
+ return;
223
+ }
224
+ if (row.status === 'IN_PROGRESS') {
225
+ // The GENERATIVE CARVE-OUT is the one thing a handler may replace: the value the request
226
+ // settles to. The outcome was resolved at SUBMIT and PERSISTED (`_scripted_*`, stripped by
227
+ // `view`), so completion stays a pure fold of the log — this path never consults the engine.
228
+ const out = '_scripted_output' in row ? row._scripted_output : stubOutput(String(row.model_id ?? ''), row.input);
229
+ const priorLogs = row.logs ?? [];
230
+ await write('request', id, {
231
+ status: 'COMPLETED',
232
+ logs: [...priorLogs, makeLog('Inference complete', 'INFO', req.occurredAt)],
233
+ output: out,
234
+ completed_at: nowIso(req.occurredAt),
235
+ inference_time: typeof row._scripted_inference_time === 'number' ? row._scripted_inference_time : 0.42,
236
+ poll_count: pollCount + 1,
237
+ }, 'request.complete', req);
238
+ }
239
+ // COMPLETED: idempotent no-op — repeated polls of a settled request never regress state.
240
+ }
241
+ // ── submit (queue and sync surfaces share this; sync additionally drives it to completion) ────
242
+ async function handleSubmit(modelId, rawBody, surface, req) {
243
+ const parsed = parseBody(rawBody);
244
+ // The gateway's own universal validation (independent of any per-model schema, which fal's
245
+ // gateway does not itself know): the submitted input must be a JSON object. Doc-UNVERIFIED
246
+ // exact message text — the FastAPI/Pydantic `{"detail":[{loc,msg,type}]}` shape itself IS
247
+ // grounded (fal's documented 422 examples use it); this specific trigger condition is a
248
+ // plausible, defensible modeling of it, not a confirmed live-API capture.
249
+ if (!parsed.ok) {
250
+ return validation422([{ loc: ['body'], msg: 'Input should be a valid dictionary or object to extract fields from', type: 'model_attributes_type' }]);
251
+ }
252
+ const input = parsed.value;
253
+ const createdAt = nowIso(req.occurredAt);
254
+ // DETERMINISTIC (R9, resource level): the request id is a UUID-SHAPED mint over
255
+ // `${type}:${occurredAt}:${ordinal}` — the world instant the submit happened at plus the count
256
+ // of request rows already stored, read BEFORE the insert — never `randomUUID`, so two
257
+ // identical worlds mint the same id and the queue's status/result reads are byte-identical.
258
+ // Two identical submits at one instant still get distinct ids: the ordinal moved.
259
+ const ordinal = rows('request', req.root).length;
260
+ const hex = createHash('sha256').update(`request:${createdAt}:${ordinal}`).digest('hex');
261
+ const id = `${hex.slice(0, 8)}-${hex.slice(8, 12)}-4${hex.slice(13, 16)}-${((parseInt(hex[16], 16) & 0x3) | 0x8).toString(16)}${hex.slice(17, 20)}-${hex.slice(20, 32)}`;
262
+ // Scenario: resolve any scripted outcome NOW (submit time) and persist it on the resource, so
263
+ // the status progression is a pure fold of the log. A miss leaves the labeled stub in place.
264
+ const scripted = req.scenarioEngine
265
+ ? (() => {
266
+ const decision = req.scenarioEngine.next({ modelId, input, surface: surface === 'sync' ? 'sync' : 'queue', nthRequest: ordinal + 1 });
267
+ return decision.kind === 'handler' ? decision.respond : null;
268
+ })()
269
+ : null;
270
+ await write('request', id, {
271
+ model_id: modelId,
272
+ input,
273
+ ...(scripted ? { _scripted_output: scripted.output } : {}),
274
+ ...(scripted?.inferenceTime !== undefined ? { _scripted_inference_time: scripted.inferenceTime } : {}),
275
+ status: 'IN_QUEUE',
276
+ queue_position: 0,
277
+ logs: [],
278
+ output: null,
279
+ error: null,
280
+ error_type: null,
281
+ created_at: createdAt,
282
+ completed_at: null,
283
+ webhook_url: null,
284
+ poll_count: 0,
285
+ }, 'request.create', req);
286
+ if (surface === 'sync') {
287
+ // fal.run settles INLINE (no queue polling exposed) and returns the model's direct output —
288
+ // no queue envelope. Drive the SAME deterministic 3-step progression the queue surface takes
289
+ // across 3 polls (poll1 IN_QUEUE echo -> poll2 IN_PROGRESS -> poll3 COMPLETED) in one call —
290
+ // mirrors replicate's `Prefer: wait` inline-settle pattern.
291
+ await progressOnce(id, req);
292
+ await progressOnce(id, req);
293
+ await progressOnce(id, req);
294
+ const settled = getRow('request', id, req.root);
295
+ return { status: 200, body: { output: settled.output } };
296
+ }
297
+ const row = getRow('request', id, req.root);
298
+ return { status: 200, body: submitEnvelope(row, modelId, id) };
299
+ }
300
+ async function handleStatus(modelId, requestId, includeLogs, req) {
301
+ const existing = getRow('request', requestId, req.root);
302
+ if (!existing)
303
+ return detailNotFound('Request not found.');
304
+ await progressOnce(requestId, req);
305
+ const row = getRow('request', requestId, req.root);
306
+ return { status: 200, body: statusView(row, modelId, requestId, includeLogs) };
307
+ }
308
+ async function handleResult(requestId, req) {
309
+ const row = getRow('request', requestId, req.root);
310
+ if (!row)
311
+ return detailNotFound('Request not found.');
312
+ // Pre-completion behavior is genuinely unmodeled (`fal.queue.response_long_poll`, todo — this
313
+ // twin settles instantly on `status` polls rather than long-polling); every DONE capability
314
+ // that reads this route first drives the request to COMPLETED via `status` polls.
315
+ return { status: 200, body: { output: row.output ?? null } };
316
+ }
317
+ async function handleCancel(requestId, req) {
318
+ const row = getRow('request', requestId, req.root);
319
+ // Cancel's 404 is its OWN documented shape (`{"status":"NOT_FOUND"}`), NOT the generic
320
+ // `{detail}` 404 envelope every other unknown-resource route in this pack uses — grounded
321
+ // verbatim from fal.ai/docs/model-endpoints/queue's cancel-endpoint error table.
322
+ if (!row)
323
+ return { status: 404, body: { status: 'NOT_FOUND' } };
324
+ if (row.status === 'COMPLETED')
325
+ return { status: 400, body: { status: 'ALREADY_COMPLETED' } };
326
+ // The real terminal state a successful cancellation settles into is UNKNOWN (fal's documented
327
+ // status enum — IN_QUEUE/IN_PROGRESS/COMPLETED — has no CANCELLED member; `⚠4` in the build
328
+ // spec). This twin claims ONLY the grounded 202 acknowledgement body, not any resulting state
329
+ // transition (`fal.queue.cancelled_terminal_state` stays `todo`) — deliberately does NOT
330
+ // mutate `row.status`, so a subsequent status poll continues its ordinary progression rather
331
+ // than asserting an invented terminal state.
332
+ return { status: 202, body: { status: 'CANCELLATION_REQUESTED' } };
333
+ }
334
+ // ── route handler ────────────────────────────────────────────────────────────────────
335
+ export async function handleFalTwinRequest(req) {
336
+ const method = req.method.toUpperCase();
337
+ const [rawPath, rawQuery] = req.path.split('?');
338
+ const path = (rawPath ?? '/').replace(/\/+$/, '') || '/';
339
+ const headers = lowerHeaders(req.headers);
340
+ const surface = routeFalSurface({ host: req.host, headers: req.headers, path });
341
+ if (req.readOnly && method !== 'GET' && method !== 'HEAD') {
342
+ return { status: 405, body: { detail: 'twin is read-only; omit readOnly to accept writes' } };
343
+ }
344
+ if (path === '/' || path === '') {
345
+ return { status: 200, body: { service: 'fal', object: 'twin' } };
346
+ }
347
+ // ── REST surface: today, only the JWKS endpoint ──
348
+ if (surface === 'rest') {
349
+ if (method === 'GET' && path === '/.well-known/jwks.json') {
350
+ return { status: 200, body: await falJwks(req.root) };
351
+ }
352
+ return detailNotFound();
353
+ }
354
+ const seg = path.replace(/^\/+/, '').split('/').filter(Boolean);
355
+ const reqIdx = seg.indexOf('requests');
356
+ // ── submit: POST /{model_id} (model ids are 2- or 3-segment, e.g. fal-ai/flux/schnell) ──
357
+ if (reqIdx === -1) {
358
+ if (method === 'POST' && seg.length >= 1) {
359
+ return handleSubmit(seg.map(decodeURIComponent).join('/'), req.body, surface, req);
360
+ }
361
+ return detailNotFound();
362
+ }
363
+ // ── queue lifecycle: {model_id}/requests/{request_id}[/status[/stream]|/response|/cancel] ──
364
+ const modelId = seg.slice(0, reqIdx).map(decodeURIComponent).join('/');
365
+ const requestId = decodeURIComponent(seg[reqIdx + 1] ?? '');
366
+ const rest = seg.slice(reqIdx + 2);
367
+ if (method === 'GET' && rest.length === 1 && rest[0] === 'status') {
368
+ const query = new URLSearchParams(rawQuery ?? '');
369
+ const includeLogs = query.get('logs') === '1';
370
+ return handleStatus(modelId, requestId, includeLogs, req);
371
+ }
372
+ if (method === 'GET' && rest.length === 2 && rest[0] === 'status' && rest[1] === 'stream') {
373
+ // SSE streaming (`fal.streaming.status_stream_sse`, todo) — genuinely unmodeled.
374
+ return detailNotFound();
375
+ }
376
+ if (method === 'GET' && rest.length === 0) {
377
+ return handleResult(requestId, req);
378
+ }
379
+ if (method === 'GET' && rest.length === 1 && rest[0] === 'response') {
380
+ // The doc-example alias route — served so a doc-literal caller works too (⚠2).
381
+ return handleResult(requestId, req);
382
+ }
383
+ if (method === 'PUT' && rest.length === 1 && rest[0] === 'cancel') {
384
+ return handleCancel(requestId, req);
385
+ }
386
+ return detailNotFound();
387
+ }
388
+ // Endpoint inventory used by the conformance snapshot (self-referential — see fal-conformance.ts
389
+ // header note and docs/contributing/conformance.md's "2 spec" discussion of this pattern, shared with replicate/
390
+ // elevenlabs/polar). fal's real v1 surface is intentionally much smaller than replicate's (no
391
+ // models/versions/collections/trainings/deployments CRUD — build spec §0), so this snapshot
392
+ // counts each INDIVIDUALLY-DOCUMENTED request/response CONTRACT (not just each distinct URL
393
+ // pattern) as its own entry — e.g. cancel's three distinct, separately-grounded status-code
394
+ // contracts (202/400/404) are 3 entries, not 1 — since each is independently verified by its own
395
+ // `done` capability and lumping them would UNDER-count real, grounded, implemented behavior
396
+ // rather than pad it.
397
+ export function falTwinSnapshot() {
398
+ return {
399
+ resourceTypes: FAL_RESOURCE_TYPES,
400
+ implementedEndpoints: [
401
+ 'POST queue.fal.run/{model_id} (queue submit)',
402
+ 'POST fal.run/{model_id} (sync submit, direct output)',
403
+ 'GET queue.fal.run/{model_id}/requests/{request_id}/status (IN_QUEUE/IN_PROGRESS/COMPLETED progression, logs param)',
404
+ 'GET queue.fal.run/{model_id}/requests/{request_id}/status (404 unknown request_id)',
405
+ 'GET queue.fal.run/{model_id}/requests/{request_id} (result, client-built bare form)',
406
+ 'GET queue.fal.run/{model_id}/requests/{request_id}/response (result, docs-example alias form)',
407
+ 'PUT queue.fal.run/{model_id}/requests/{request_id}/cancel (202 CANCELLATION_REQUESTED)',
408
+ 'PUT queue.fal.run/{model_id}/requests/{request_id}/cancel (400 ALREADY_COMPLETED)',
409
+ 'PUT queue.fal.run/{model_id}/requests/{request_id}/cancel (404 NOT_FOUND)',
410
+ 'POST queue.fal.run/{model_id} (422 validation error)',
411
+ 'GET rest.fal.ai/.well-known/jwks.json',
412
+ ],
413
+ };
414
+ }