graphile-presigned-url-plugin 1.11.4 → 1.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,262 @@
1
+ /**
2
+ * The managed upload lifecycle, shared by both transports.
3
+ *
4
+ * Multipart-through-GraphQL and presigned two-step are transports for one
5
+ * lifecycle, not two data models: either way an object gets a files row, a
6
+ * server-chosen key, a tenant-resolved bucket, and a projection document that
7
+ * carries the files row's id. The only difference is who moves the bytes.
8
+ *
9
+ * This module owns the parts that are the same:
10
+ * * `resolveManagedUploadTarget` — from a document column to a concrete
11
+ * (storage module, bucket, physical bucket, S3 config).
12
+ * * `finalizeStagedUpload` — from bytes already in S3 under a staging key to a
13
+ * files row and a projection document, deduplicating on content hash.
14
+ * * `buildFileProjection` — the document shape the column stores.
15
+ *
16
+ * Nothing here bakes a presigned URL into a row: `url` is populated only for a
17
+ * public bucket, where it is a stable CDN address rather than a credential with
18
+ * an expiry.
19
+ */
20
+ import { Logger } from '@pgpmjs/logger';
21
+ import { resolveDefaultBucket } from './default-bucket';
22
+ import { getFileRefFieldBinding } from './file-ref-registry';
23
+ import { provisionAndRecordPhysicalBucket, resolveS3ForDatabase } from './physical-bucket';
24
+ import { withRequestPgClient } from './request-pg-client';
25
+ import { copyS3Object, deleteS3Object } from './s3-signer';
26
+ import { getBucketConfig, loadAllStorageModules } from './storage-module-cache';
27
+ const log = new Logger('graphile-presigned-url:managed-upload');
28
+ /**
29
+ * Build the projection document for a files row.
30
+ *
31
+ * A public bucket has a stable address, so `url` is a real, durable value there.
32
+ * A private bucket has no such address — only presigned, expiring ones — so the
33
+ * field is omitted rather than filled with a URL that dies in an hour.
34
+ */
35
+ export function buildFileProjection(file, bucket, s3) {
36
+ const projection = {
37
+ id: file.id,
38
+ key: file.key,
39
+ bucket_id: file.bucketId,
40
+ mime: file.mime,
41
+ size: file.size,
42
+ };
43
+ if (file.filename)
44
+ projection.filename = file.filename;
45
+ if (bucket.is_public && s3.publicUrlPrefix) {
46
+ projection.url = `${s3.publicUrlPrefix.replace(/\/$/, '')}/${file.key}`;
47
+ }
48
+ return projection;
49
+ }
50
+ /**
51
+ * Resolve where a write to a document column lands.
52
+ *
53
+ * Two routes, one rule — the bucket is always resolved inside the tenant:
54
+ * * a registered column names its storage module, and either a logical bucket
55
+ * key or the reserved default tag for its declared publicness;
56
+ * * an unregistered column (a bare `image`/`upload` on a database provisioned
57
+ * before the registry) falls back to the app-scope module and the same
58
+ * reserved default tag. That is a *tenant* default, not an environment one.
59
+ *
60
+ * A database with no storage module raises: there is nowhere tenant-owned to put
61
+ * the bytes, and the deployment's configured bucket is not an answer.
62
+ */
63
+ export async function resolveManagedUploadTarget(args) {
64
+ const { options, withPgClient, pgSettings, databaseId, field, defaultPublicAccess } = args;
65
+ // Registry and module registration are schema metadata, not tenant rows:
66
+ // read them in the system lane, like every other config read here.
67
+ const binding = await withPgClient(null, async (pgClient) => {
68
+ try {
69
+ return await getFileRefFieldBinding(pgClient, databaseId, field);
70
+ }
71
+ catch (err) {
72
+ // An unregistered column is a legitimate state (it predates the registry)
73
+ // and falls back to the tenant's app-scope default below. Any other
74
+ // failure — a broken connection, a missing registry table — is not.
75
+ if (err?.name === 'FileRefFieldNotRegisteredError')
76
+ return null;
77
+ throw err;
78
+ }
79
+ });
80
+ const allConfigs = await withPgClient(null, (pgClient) => loadAllStorageModules(pgClient, databaseId));
81
+ const storageConfig = binding
82
+ ? allConfigs.find((c) => c.id === binding.storageModuleId)
83
+ : allConfigs.find((c) => c.scope === 'app');
84
+ if (!storageConfig) {
85
+ throw new Error(binding
86
+ ? `STORAGE_MODULE_NOT_FOUND: file_ref_field ${binding.id} names storage module ` +
87
+ `${binding.storageModuleId}, which database ${databaseId} does not have`
88
+ : `STORAGE_MODULE_NOT_FOUND: ${field.schemaName}.${field.tableName}.${field.columnName} is an ` +
89
+ `unregistered upload column and database ${databaseId} has no app-scope storage module to ` +
90
+ 'default to; there is no environment bucket to fall back to');
91
+ }
92
+ if (storageConfig.scope !== 'app') {
93
+ // An entity-scoped module resolves its bucket per owning row, and a
94
+ // multipart column write does not carry one. Refuse rather than write a
95
+ // tenant's file into whichever bucket happened to resolve.
96
+ throw new Error(`STORAGE_SCOPE_UNSUPPORTED: ${field.schemaName}.${field.tableName}.${field.columnName} binds to ` +
97
+ `'${storageConfig.scope}'-scoped storage, which resolves its bucket per owner row. ` +
98
+ 'Use the presigned upload mutation, which takes an ownerId.');
99
+ }
100
+ const publicAccess = binding?.isPublic ?? defaultPublicAccess;
101
+ // Bucket resolution and the bucket read run under the request role: what the
102
+ // caller may store into is exactly what RLS lets them see.
103
+ const coordinate = await withRequestPgClient(withPgClient, pgSettings, (pgClient) => resolveDefaultBucket(pgClient, databaseId, storageConfig.scope, null, publicAccess, binding?.bucketKey ?? null));
104
+ const bucket = await withRequestPgClient(withPgClient, pgSettings, (pgClient) => getBucketConfig(pgClient, storageConfig, databaseId, coordinate.resolvedKey));
105
+ if (!bucket) {
106
+ throw new Error(`BUCKET_NOT_FOUND: bucket "${coordinate.resolvedKey}" resolved for ` +
107
+ `${field.schemaName}.${field.tableName}.${field.columnName} is not readable`);
108
+ }
109
+ if (bucket.allow_custom_keys) {
110
+ // A path-keyed bucket (e.g. a static site's) is addressed by the keys its
111
+ // publisher chose; this lane can only mint content-hash keys, which would
112
+ // pollute it with unreachable objects. Path-keyed uploads go through the
113
+ // presigned lane, which accepts an explicit `key`.
114
+ throw new Error(`BUCKET_PATH_KEYED: bucket "${bucket.key}" allows custom keys and is addressed by path; ` +
115
+ 'the multipart upload lane only writes content-addressed keys. Use the presigned upload ' +
116
+ 'mutation with an explicit key.');
117
+ }
118
+ const physicalName = bucket.physical_name === null
119
+ ? await provisionAndRecordPhysicalBucket(options, withPgClient, storageConfig, databaseId, bucket, storageConfig.allowedOrigins)
120
+ : bucket.physical_name;
121
+ return {
122
+ databaseId,
123
+ storageConfig,
124
+ bucket,
125
+ physicalName,
126
+ s3: resolveS3ForDatabase(options, storageConfig, physicalName),
127
+ binding,
128
+ };
129
+ }
130
+ /**
131
+ * Validate an upload against the resolved bucket's rules.
132
+ *
133
+ * The same rules the presigned lane enforces — a transport must not be a way
134
+ * around a bucket's mime allowlist or size cap.
135
+ */
136
+ export function assertUploadAllowedByBucket(target, contentType, size) {
137
+ const { bucket, storageConfig } = target;
138
+ if (bucket.allowed_mime_types && bucket.allowed_mime_types.length > 0) {
139
+ const isAllowed = bucket.allowed_mime_types.some((pattern) => {
140
+ if (pattern === '*/*')
141
+ return true;
142
+ if (pattern.endsWith('/*'))
143
+ return contentType.startsWith(pattern.slice(0, -1));
144
+ return contentType === pattern;
145
+ });
146
+ if (!isAllowed) {
147
+ throw new Error(`CONTENT_TYPE_NOT_ALLOWED: ${contentType} not in bucket allowed types`);
148
+ }
149
+ }
150
+ const maxSize = bucket.max_file_size ?? storageConfig.defaultMaxFileSize;
151
+ if (size > maxSize) {
152
+ throw new Error(`FILE_TOO_LARGE: ${size} bytes exceeds the ${maxSize} byte limit`);
153
+ }
154
+ if (size <= 0) {
155
+ throw new Error('INVALID_FILE_SIZE: an upload must carry at least one byte');
156
+ }
157
+ }
158
+ /** Delete an object we are abandoning, without masking the failure in progress. */
159
+ async function bestEffortDelete(s3, key) {
160
+ try {
161
+ await deleteS3Object(s3, key);
162
+ }
163
+ catch (err) {
164
+ log.warn(`Failed to clean up abandoned object ${key}: ${err}`);
165
+ }
166
+ }
167
+ /**
168
+ * Turn bytes already staged in S3 into a files row and a projection document.
169
+ *
170
+ * The content hash is only known once the stream has been read, so a streaming
171
+ * transport writes to a staging key first and promotes here:
172
+ *
173
+ * * hash already present in this bucket → drop the staged object, reuse the
174
+ * existing files row. Dedup is a property of the object, so it holds no
175
+ * matter which transport wrote it first.
176
+ * * otherwise → server-side copy to the content-addressed key, drop the staged
177
+ * object, insert the files row.
178
+ *
179
+ * The row is inserted *after* the bytes land, so the confirm-upload job the
180
+ * insert trigger enqueues finds the object and completes the
181
+ * `requested → uploaded` transition without any extra wiring here.
182
+ *
183
+ * Every failure path leaves S3 as it found it. Bytes written by this call and
184
+ * not reachable through a files row would be invisible to storage GC, which
185
+ * collects objects by walking rows — so an object is only left behind once the
186
+ * row naming it exists.
187
+ */
188
+ export async function finalizeStagedUpload(args) {
189
+ const { target, withPgClient, pgSettings, staged } = args;
190
+ const { storageConfig, bucket, s3 } = target;
191
+ assertUploadAllowedByBucket(target, staged.contentType, staged.size);
192
+ const finalKey = staged.contentHash;
193
+ const existing = await withRequestPgClient(withPgClient, pgSettings, async (pgClient) => {
194
+ const result = await pgClient.query({
195
+ text: `SELECT id, key, mime_type, size, filename
196
+ FROM ${storageConfig.filesQualifiedName}
197
+ WHERE content_hash = $1 AND bucket_id = $2
198
+ LIMIT 1`,
199
+ values: [staged.contentHash, bucket.id],
200
+ });
201
+ return result.rows[0];
202
+ });
203
+ if (existing) {
204
+ log.info(`Dedup hit: file ${existing.id} already carries hash ${staged.contentHash}`);
205
+ await deleteS3Object(s3, staged.stagingKey);
206
+ return {
207
+ projection: buildFileProjection({
208
+ id: existing.id,
209
+ key: existing.key,
210
+ bucketId: bucket.id,
211
+ mime: existing.mime_type,
212
+ size: Number(existing.size),
213
+ filename: existing.filename,
214
+ }, bucket, s3),
215
+ deduplicated: true,
216
+ };
217
+ }
218
+ await copyS3Object(s3, staged.stagingKey, finalKey, staged.contentType);
219
+ let fileId;
220
+ try {
221
+ fileId = await withRequestPgClient(withPgClient, pgSettings, async (pgClient) => {
222
+ const result = await pgClient.query({
223
+ text: `INSERT INTO ${storageConfig.filesQualifiedName}
224
+ (bucket_id, key, content_hash, mime_type, size, filename, is_public)
225
+ VALUES ($1, $2, $3, $4, $5, $6, $7)
226
+ RETURNING id`,
227
+ values: [
228
+ bucket.id,
229
+ finalKey,
230
+ staged.contentHash,
231
+ staged.contentType,
232
+ staged.size,
233
+ staged.filename ?? null,
234
+ bucket.is_public,
235
+ ],
236
+ });
237
+ return result.rows[0].id;
238
+ });
239
+ }
240
+ catch (err) {
241
+ // No row names either key, so both are unreachable to GC. Dropping the
242
+ // promoted copy is safe precisely because the dedup probe above found no row
243
+ // on this hash: nothing else in this bucket is entitled to those bytes.
244
+ // Cleanup must never replace the failure that caused it.
245
+ await bestEffortDelete(s3, finalKey);
246
+ await bestEffortDelete(s3, staged.stagingKey);
247
+ throw err;
248
+ }
249
+ await deleteS3Object(s3, staged.stagingKey);
250
+ log.info(`Managed upload created file ${fileId} at ${bucket.key}/${finalKey}`);
251
+ return {
252
+ projection: buildFileProjection({
253
+ id: fileId,
254
+ key: finalKey,
255
+ bucketId: bucket.id,
256
+ mime: staged.contentType,
257
+ size: staged.size,
258
+ filename: staged.filename,
259
+ }, bucket, s3),
260
+ deduplicated: false,
261
+ };
262
+ }
@@ -0,0 +1,60 @@
1
+ /**
2
+ * Physical bucket coordinates: minting a name once, recording it, and building
3
+ * an S3 config against a *known* name.
4
+ *
5
+ * A logical bucket belongs to a tenant; a physical bucket is an S3 name. The
6
+ * mapping is recorded on the bucket row the first time it is provisioned, and
7
+ * from then on the recorded value is the only coordinate anything reads — no
8
+ * name is ever recomputed from a prefix convention, and there is no
9
+ * environment-level bucket standing in for a tenant's.
10
+ */
11
+ import { type WithPgClient } from './request-pg-client';
12
+ import type { BucketConfig, PresignedUrlPluginOptions, S3Config, StorageModuleConfig } from './types';
13
+ /**
14
+ * Resolve the plugin's S3 connection (credentials, endpoint, region), memoizing
15
+ * a lazy getter on first use.
16
+ *
17
+ * `s3.bucket` on the result is the deployment's *default* physical bucket. It is
18
+ * a connection default only — never a tenant's bucket. Every upload path
19
+ * resolves its physical bucket from the tenant's bucket row.
20
+ */
21
+ export declare function resolveS3(options: PresignedUrlPluginOptions): S3Config;
22
+ /**
23
+ * Mint the physical S3 bucket name for a logical bucket's first provision.
24
+ *
25
+ * This is a naming *policy*, consulted exactly once per bucket — before the
26
+ * physical bucket exists. Once provisioned, the recorded `physical_name` on the
27
+ * row is authoritative and this function must not be consulted again.
28
+ *
29
+ * There is no fallback to the configured `s3.bucket`: a deployment-wide bucket
30
+ * name is not a tenant's storage, and silently minting one is how objects ended
31
+ * up in a bucket no database owned. A deployment that wants per-tenant buckets
32
+ * must supply the policy.
33
+ */
34
+ export declare function mintPhysicalBucketName(options: PresignedUrlPluginOptions, databaseId: string, bucketKey: string): string;
35
+ /**
36
+ * Build the S3 config for a *known* physical bucket. `physicalName` is
37
+ * required — callers must resolve the coordinate (stored row value, or a
38
+ * freshly provisioned name) before getting here. No name is ever recomputed.
39
+ */
40
+ export declare function resolveS3ForDatabase(options: PresignedUrlPluginOptions, storageConfig: StorageModuleConfig, physicalName: string): S3Config;
41
+ /**
42
+ * First provision of a logical bucket: mint a name, create the physical S3
43
+ * bucket, and record the exact name on the source row. Returns the recorded
44
+ * physical name.
45
+ *
46
+ * Only called when the row has no `physical_name` yet. Afterwards the stored
47
+ * value is the durable coordinate: route resolution and every later read use
48
+ * it verbatim; nothing is recomputed.
49
+ *
50
+ * The record write runs in the system lane (privileged role, so it bypasses the
51
+ * RLS that stops request roles from UPDATE-ing bucket rows) — it is server
52
+ * bookkeeping, not request data. It still carries the tenant `database_id`
53
+ * claim, because the buckets table's catalog-sync trigger calls
54
+ * `jwt_private.current_database_id()` and would otherwise raise
55
+ * DATABASE_CLAIM_REQUIRED; `withRequestPgClient` applies that claim inside the
56
+ * write's transaction without switching off the privileged role.
57
+ * `bucket` (the cached config) is mutated in place so subsequent reads observe
58
+ * the recorded name without a DB round-trip.
59
+ */
60
+ export declare function provisionAndRecordPhysicalBucket(options: PresignedUrlPluginOptions, withPgClient: WithPgClient, storageConfig: StorageModuleConfig, databaseId: string, bucket: BucketConfig, allowedOrigins: string[] | null): Promise<string>;
@@ -0,0 +1,111 @@
1
+ /**
2
+ * Physical bucket coordinates: minting a name once, recording it, and building
3
+ * an S3 config against a *known* name.
4
+ *
5
+ * A logical bucket belongs to a tenant; a physical bucket is an S3 name. The
6
+ * mapping is recorded on the bucket row the first time it is provisioned, and
7
+ * from then on the recorded value is the only coordinate anything reads — no
8
+ * name is ever recomputed from a prefix convention, and there is no
9
+ * environment-level bucket standing in for a tenant's.
10
+ */
11
+ import { Logger } from '@pgpmjs/logger';
12
+ import { withRequestPgClient } from './request-pg-client';
13
+ import { isS3BucketProvisioned, markS3BucketProvisioned } from './storage-module-cache';
14
+ const log = new Logger('graphile-presigned-url:physical-bucket');
15
+ /**
16
+ * Resolve the plugin's S3 connection (credentials, endpoint, region), memoizing
17
+ * a lazy getter on first use.
18
+ *
19
+ * `s3.bucket` on the result is the deployment's *default* physical bucket. It is
20
+ * a connection default only — never a tenant's bucket. Every upload path
21
+ * resolves its physical bucket from the tenant's bucket row.
22
+ */
23
+ export function resolveS3(options) {
24
+ if (typeof options.s3 === 'function') {
25
+ const resolved = options.s3();
26
+ options.s3 = resolved;
27
+ return resolved;
28
+ }
29
+ return options.s3;
30
+ }
31
+ /**
32
+ * Mint the physical S3 bucket name for a logical bucket's first provision.
33
+ *
34
+ * This is a naming *policy*, consulted exactly once per bucket — before the
35
+ * physical bucket exists. Once provisioned, the recorded `physical_name` on the
36
+ * row is authoritative and this function must not be consulted again.
37
+ *
38
+ * There is no fallback to the configured `s3.bucket`: a deployment-wide bucket
39
+ * name is not a tenant's storage, and silently minting one is how objects ended
40
+ * up in a bucket no database owned. A deployment that wants per-tenant buckets
41
+ * must supply the policy.
42
+ */
43
+ export function mintPhysicalBucketName(options, databaseId, bucketKey) {
44
+ if (!options.resolveBucketName) {
45
+ throw new Error('STORAGE_BUCKET_NAME_POLICY_MISSING: no resolveBucketName was configured, so there is ' +
46
+ `no name to provision for bucket "${bucketKey}" of database ${databaseId}. ` +
47
+ 'Physical bucket naming is a deployment policy; the configured s3.bucket is a ' +
48
+ 'connection default and is never a tenant bucket.');
49
+ }
50
+ return options.resolveBucketName(databaseId, bucketKey);
51
+ }
52
+ /**
53
+ * Build the S3 config for a *known* physical bucket. `physicalName` is
54
+ * required — callers must resolve the coordinate (stored row value, or a
55
+ * freshly provisioned name) before getting here. No name is ever recomputed.
56
+ */
57
+ export function resolveS3ForDatabase(options, storageConfig, physicalName) {
58
+ const globalS3 = resolveS3(options);
59
+ const publicUrlPrefix = storageConfig.publicUrlPrefix != null
60
+ ? storageConfig.publicUrlPrefix
61
+ : globalS3.publicUrlPrefix;
62
+ if (physicalName === globalS3.bucket && publicUrlPrefix === globalS3.publicUrlPrefix) {
63
+ return globalS3;
64
+ }
65
+ return {
66
+ ...globalS3,
67
+ bucket: physicalName,
68
+ ...(publicUrlPrefix != null ? { publicUrlPrefix } : {}),
69
+ };
70
+ }
71
+ /**
72
+ * First provision of a logical bucket: mint a name, create the physical S3
73
+ * bucket, and record the exact name on the source row. Returns the recorded
74
+ * physical name.
75
+ *
76
+ * Only called when the row has no `physical_name` yet. Afterwards the stored
77
+ * value is the durable coordinate: route resolution and every later read use
78
+ * it verbatim; nothing is recomputed.
79
+ *
80
+ * The record write runs in the system lane (privileged role, so it bypasses the
81
+ * RLS that stops request roles from UPDATE-ing bucket rows) — it is server
82
+ * bookkeeping, not request data. It still carries the tenant `database_id`
83
+ * claim, because the buckets table's catalog-sync trigger calls
84
+ * `jwt_private.current_database_id()` and would otherwise raise
85
+ * DATABASE_CLAIM_REQUIRED; `withRequestPgClient` applies that claim inside the
86
+ * write's transaction without switching off the privileged role.
87
+ * `bucket` (the cached config) is mutated in place so subsequent reads observe
88
+ * the recorded name without a DB round-trip.
89
+ */
90
+ export async function provisionAndRecordPhysicalBucket(options, withPgClient, storageConfig, databaseId, bucket, allowedOrigins) {
91
+ const s3BucketName = mintPhysicalBucketName(options, databaseId, bucket.key);
92
+ if (options.ensureBucketProvisioned && !isS3BucketProvisioned(s3BucketName)) {
93
+ log.info(`Lazy-provisioning S3 bucket "${s3BucketName}" for database ${databaseId}`);
94
+ await options.ensureBucketProvisioned(s3BucketName, bucket.type, databaseId, allowedOrigins);
95
+ markS3BucketProvisioned(s3BucketName);
96
+ log.info(`Lazy-provisioned S3 bucket "${s3BucketName}" successfully`);
97
+ }
98
+ // Record the physical coordinate on the source row. The `physical_name IS NULL`
99
+ // guard keeps this idempotent and race-safe across concurrent first uploads.
100
+ // The catalog-sync trigger on this UPDATE needs `jwt.claims.database_id`, so the
101
+ // write runs under the resolved database claim (privileged role preserved).
102
+ await withRequestPgClient(withPgClient, { 'jwt.claims.database_id': databaseId }, (client) => client.query({
103
+ text: `UPDATE ${storageConfig.bucketsQualifiedName}
104
+ SET physical_name = $1
105
+ WHERE id = $2 AND physical_name IS NULL`,
106
+ values: [s3BucketName, bucket.id],
107
+ }));
108
+ bucket.physical_name = s3BucketName;
109
+ log.info(`Recorded physical_name="${s3BucketName}" on bucket ${bucket.id}`);
110
+ return s3BucketName;
111
+ }