blogwright-analytics 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +162 -0
- package/dist/adapters/duckdb-ingest.d.ts +76 -0
- package/dist/adapters/duckdb-ingest.js +173 -0
- package/dist/adapters/duckdb-query.d.ts +56 -0
- package/dist/adapters/duckdb-query.js +80 -0
- package/dist/adapters/duckdb-session.d.ts +168 -0
- package/dist/adapters/duckdb-session.js +330 -0
- package/dist/app/_app/immutable/assets/0.BTQrrh5B.css +1 -0
- package/dist/app/_app/immutable/assets/2.CZSK3rT8.css +1 -0
- package/dist/app/_app/immutable/assets/BrushContext.D7c8UPey.css +1 -0
- package/dist/app/_app/immutable/assets/ChartAnnotations.CPxIG7Mw.css +1 -0
- package/dist/app/_app/immutable/assets/Circle.C5MKzgk2.css +1 -0
- package/dist/app/_app/immutable/assets/DefaultTooltip.C5-uctZ7.css +1 -0
- package/dist/app/_app/immutable/assets/Group.DV48xipa.css +1 -0
- package/dist/app/_app/immutable/assets/Labels.BxZ4NUVz.css +1 -0
- package/dist/app/_app/immutable/assets/Legend.CxnrE4Ye.css +1 -0
- package/dist/app/_app/immutable/assets/Line.fkmsECm9.css +1 -0
- package/dist/app/_app/immutable/assets/Path.CvpwNZ6g.css +1 -0
- package/dist/app/_app/immutable/assets/Rect.CtRaGMmQ.css +1 -0
- package/dist/app/_app/immutable/assets/Text.j9l35qB0.css +1 -0
- package/dist/app/_app/immutable/assets/TransformContext.Bs_HkpAk.css +1 -0
- package/dist/app/_app/immutable/assets/Voronoi.ce7atosu.css +1 -0
- package/dist/app/_app/immutable/chunks/-aNGNaBT.js +1 -0
- package/dist/app/_app/immutable/chunks/6djn-yLs.js +1 -0
- package/dist/app/_app/immutable/chunks/B1amyutE.js +1 -0
- package/dist/app/_app/immutable/chunks/B3vZDoek.js +1 -0
- package/dist/app/_app/immutable/chunks/B5KRA4hC.js +1 -0
- package/dist/app/_app/immutable/chunks/BClnVG6H.js +1 -0
- package/dist/app/_app/immutable/chunks/BID1NNRh.js +1 -0
- package/dist/app/_app/immutable/chunks/BR2LaRms.js +1 -0
- package/dist/app/_app/immutable/chunks/Bd1gDe3Y.js +1 -0
- package/dist/app/_app/immutable/chunks/Bjy-W4x2.js +81 -0
- package/dist/app/_app/immutable/chunks/Bl052uUt.js +1 -0
- package/dist/app/_app/immutable/chunks/Bye3lL0c.js +1 -0
- package/dist/app/_app/immutable/chunks/C58PZtCD.js +4 -0
- package/dist/app/_app/immutable/chunks/CAzydqEO.js +1 -0
- package/dist/app/_app/immutable/chunks/CCch3uox.js +1 -0
- package/dist/app/_app/immutable/chunks/CIlSMUH9.js +1 -0
- package/dist/app/_app/immutable/chunks/CO1vUXfR.js +1 -0
- package/dist/app/_app/immutable/chunks/CPbD8C65.js +5 -0
- package/dist/app/_app/immutable/chunks/CRTcXoMo.js +1 -0
- package/dist/app/_app/immutable/chunks/CjjyIQAO.js +1 -0
- package/dist/app/_app/immutable/chunks/CuXAxjvF.js +1 -0
- package/dist/app/_app/immutable/chunks/CvyVA_jC.js +1 -0
- package/dist/app/_app/immutable/chunks/CxGCFVdy.js +1 -0
- package/dist/app/_app/immutable/chunks/D0Ty6LN0.js +1 -0
- package/dist/app/_app/immutable/chunks/D2AaQUUW.js +1 -0
- package/dist/app/_app/immutable/chunks/D2BnX0Uk.js +3 -0
- package/dist/app/_app/immutable/chunks/DJc8C0NK.js +1 -0
- package/dist/app/_app/immutable/chunks/DKMlMI4a.js +1 -0
- package/dist/app/_app/immutable/chunks/DVXZkpbf.js +1 -0
- package/dist/app/_app/immutable/chunks/DVt8ukQ_.js +1 -0
- package/dist/app/_app/immutable/chunks/DZPlYdq_.js +1 -0
- package/dist/app/_app/immutable/chunks/Db0q5_zr.js +1 -0
- package/dist/app/_app/immutable/chunks/Dfvzj6n2.js +1 -0
- package/dist/app/_app/immutable/chunks/Dh958be7.js +1 -0
- package/dist/app/_app/immutable/chunks/DjKLLdnY.js +15 -0
- package/dist/app/_app/immutable/chunks/Doz7YX1W.js +1 -0
- package/dist/app/_app/immutable/chunks/DthYhn6Y.js +2 -0
- package/dist/app/_app/immutable/chunks/DtuTIrAM.js +1 -0
- package/dist/app/_app/immutable/chunks/HclGiUj8.js +1 -0
- package/dist/app/_app/immutable/chunks/Hx0TNsV3.js +1 -0
- package/dist/app/_app/immutable/chunks/RobXhXPM.js +1 -0
- package/dist/app/_app/immutable/chunks/V9ZjaxiY.js +1 -0
- package/dist/app/_app/immutable/chunks/Y5urAfNy.js +1 -0
- package/dist/app/_app/immutable/chunks/caXkbKD3.js +1 -0
- package/dist/app/_app/immutable/chunks/devYm2ud.js +1 -0
- package/dist/app/_app/immutable/chunks/mtZWP0zR.js +1 -0
- package/dist/app/_app/immutable/chunks/vDgBJUjM.js +1 -0
- package/dist/app/_app/immutable/chunks/xIq_fFFM.js +1 -0
- package/dist/app/_app/immutable/chunks/xihTtKlq.js +1 -0
- package/dist/app/_app/immutable/chunks/z05MoCFz.js +1 -0
- package/dist/app/_app/immutable/entry/app.CLAerUAN.js +2 -0
- package/dist/app/_app/immutable/entry/start.D3MqnNci.js +1 -0
- package/dist/app/_app/immutable/nodes/0.UTMEigHJ.js +1 -0
- package/dist/app/_app/immutable/nodes/1.Cn4f11bT.js +1 -0
- package/dist/app/_app/immutable/nodes/2.B39cIcr2.js +6 -0
- package/dist/app/_app/version.json +1 -0
- package/dist/app/index.html +82 -0
- package/dist/aws/clients.d.ts +70 -0
- package/dist/aws/clients.js +52 -0
- package/dist/aws/errors.d.ts +41 -0
- package/dist/aws/errors.js +70 -0
- package/dist/aws/firehose.d.ts +228 -0
- package/dist/aws/firehose.js +347 -0
- package/dist/aws/glue.d.ts +103 -0
- package/dist/aws/glue.js +225 -0
- package/dist/aws/lambda.d.ts +132 -0
- package/dist/aws/lambda.js +339 -0
- package/dist/aws/s3tables.d.ts +120 -0
- package/dist/aws/s3tables.js +281 -0
- package/dist/backfill.d.ts +100 -0
- package/dist/backfill.js +294 -0
- package/dist/commands.d.ts +124 -0
- package/dist/commands.js +336 -0
- package/dist/config.d.ts +162 -0
- package/dist/config.js +317 -0
- package/dist/fixture-ingest.d.ts +49 -0
- package/dist/fixture-ingest.js +43 -0
- package/dist/fixture-query.d.ts +39 -0
- package/dist/fixture-query.js +70 -0
- package/dist/index.d.ts +35 -0
- package/dist/index.js +35 -0
- package/dist/nodes.d.ts +404 -0
- package/dist/nodes.js +2708 -0
- package/dist/paths.d.ts +45 -0
- package/dist/paths.js +47 -0
- package/dist/plugin.d.ts +102 -0
- package/dist/plugin.js +248 -0
- package/dist/ports.d.ts +113 -0
- package/dist/ports.js +35 -0
- package/dist/queries.d.ts +301 -0
- package/dist/queries.js +414 -0
- package/dist/schema.d.ts +240 -0
- package/dist/schema.js +154 -0
- package/dist/server.d.ts +150 -0
- package/dist/server.js +499 -0
- package/dist/transform/bots.d.ts +47 -0
- package/dist/transform/bots.js +73 -0
- package/dist/transform/handler.d.ts +135 -0
- package/dist/transform/handler.js +177 -0
- package/dist/transform/map-record.d.ts +110 -0
- package/dist/transform/map-record.js +275 -0
- package/dist/transform/visitor-key.d.ts +83 -0
- package/dist/transform/visitor-key.js +120 -0
- package/dist/transform-bundle/index.mjs +21456 -0
- package/dist/transform-bundle/transform-manifest.json +4 -0
- package/dist/transform-hash.d.ts +135 -0
- package/dist/transform-hash.js +186 -0
- package/dist/write-transform-manifest.mjs +365 -0
- package/package.json +59 -0
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
import { AwsError } from 'blogwright-core';
|
|
2
|
+
import { rethrowWithContext } from './errors.js';
|
|
3
|
+
/**
|
|
4
|
+
* Amazon S3 Tables control-plane client - create/get/delete for table buckets,
|
|
5
|
+
* namespaces and tables (the `s3tables` API, REST-JSON). It lives in
|
|
6
|
+
* `blogwright-analytics`, not in core: core's `SIGNING_NAMES` gains no `s3tables`
|
|
7
|
+
* key, and every request signs through the `{ service: 's3tables', signingName:
|
|
8
|
+
* 's3tables' }` descriptor the plugin transport seam accepts (see
|
|
9
|
+
* `packages/core/src/aws/endpoint.ts`'s `ServiceDescriptor`), which resolves to
|
|
10
|
+
* the canonical `s3tables.<region>.amazonaws.com` host. Operation names, methods
|
|
11
|
+
* and URI templates below follow the published S3 Tables API reference - GetTable
|
|
12
|
+
* is the one operation that is not path-templated like its siblings; it takes its
|
|
13
|
+
* identifiers as query parameters (`/get-table?tableBucketARN=&namespace=&name=`).
|
|
14
|
+
* The floci emulator does not implement this service, so it is covered by
|
|
15
|
+
* transport mocks in tests.
|
|
16
|
+
*
|
|
17
|
+
* `createTable`'s `schema` parameter mirrors `CreateTableRequest.metadata.iceberg`
|
|
18
|
+
* (`TableMetadata` is a union whose sole member today is `iceberg: IcebergMetadata`).
|
|
19
|
+
* Field names below are verified against the service's `IcebergMetadata`,
|
|
20
|
+
* `IcebergSchema`, `SchemaField`, `IcebergPartitionSpec` and `IcebergPartitionField`
|
|
21
|
+
* shapes: `IcebergMetadata.schema.fields` (not `schemaV2`, which exists only for
|
|
22
|
+
* nested/complex Iceberg types this table never uses) carries `{ name, type, id,
|
|
23
|
+
* required }` - already camelCase on the wire - while `IcebergPartitionSpec.fields`
|
|
24
|
+
* carries `IcebergPartitionField`, whose `source-id` and `field-id` are genuinely
|
|
25
|
+
* hyphenated JSON keys, not a documentation typo. This client accepts `sourceId`/
|
|
26
|
+
* `fieldId` (idiomatic TypeScript) and translates to the wire's hyphenated keys
|
|
27
|
+
* itself - that translation is this client's concern, not its callers'. A schema
|
|
28
|
+
* field's `id` is optional in the API (auto-assigned when omitted) but required
|
|
29
|
+
* here: a partition field can only reference an id the caller already knows, and
|
|
30
|
+
* schema and partition spec travel in the same `CreateTable` request, so an
|
|
31
|
+
* auto-assigned id would not exist yet for `sourceId` to reference.
|
|
32
|
+
*/
|
|
33
|
+
const SERVICE = { service: 's3tables', signingName: 's3tables' };
|
|
34
|
+
/** The only table format S3 Tables accepts today; named so `createTable` never repeats the literal. */
|
|
35
|
+
const ICEBERG_FORMAT = 'ICEBERG';
|
|
36
|
+
const PATHS = {
|
|
37
|
+
buckets: '/buckets',
|
|
38
|
+
bucket: (tableBucketArn) => `/buckets/${encodeURIComponent(tableBucketArn)}`,
|
|
39
|
+
namespaces: (tableBucketArn) => `/namespaces/${encodeURIComponent(tableBucketArn)}`,
|
|
40
|
+
namespace: (tableBucketArn, namespace) => `/namespaces/${encodeURIComponent(tableBucketArn)}/${encodeURIComponent(namespace)}`,
|
|
41
|
+
tables: (tableBucketArn, namespace) => `/tables/${encodeURIComponent(tableBucketArn)}/${encodeURIComponent(namespace)}`,
|
|
42
|
+
table: (tableBucketArn, namespace, name) => `/tables/${encodeURIComponent(tableBucketArn)}/${encodeURIComponent(namespace)}/${encodeURIComponent(name)}`,
|
|
43
|
+
getTable: '/get-table',
|
|
44
|
+
};
|
|
45
|
+
function normalizeTableBucket(res) {
|
|
46
|
+
return { arn: res.arn ?? '', name: res.name ?? '' };
|
|
47
|
+
}
|
|
48
|
+
// GetNamespace's response carries no `tableBucketARN` field (only the response of
|
|
49
|
+
// CreateNamespace does), so the bucket ARN comes from what the caller already
|
|
50
|
+
// passed in, not from the response.
|
|
51
|
+
function normalizeNamespace(res, tableBucketArn, fallback) {
|
|
52
|
+
return { name: res.namespace?.[0] ?? fallback, tableBucketArn };
|
|
53
|
+
}
|
|
54
|
+
function normalizeTable(res, fallbackName) {
|
|
55
|
+
return {
|
|
56
|
+
arn: res.tableARN ?? '',
|
|
57
|
+
name: res.name ?? fallbackName,
|
|
58
|
+
metadataLocation: res.metadataLocation,
|
|
59
|
+
};
|
|
60
|
+
}
|
|
61
|
+
/** Translate the client's camelCase `IcebergTableSchema` into the wire's `IcebergMetadata` shape - the one place `source-id`/`field-id`'s hyphenated JSON keys are spelled out. */
|
|
62
|
+
function buildIcebergMetadata(schema) {
|
|
63
|
+
return {
|
|
64
|
+
iceberg: {
|
|
65
|
+
schema: {
|
|
66
|
+
fields: schema.fields.map((f) => ({
|
|
67
|
+
name: f.name,
|
|
68
|
+
type: f.type,
|
|
69
|
+
id: f.id,
|
|
70
|
+
...(f.required !== undefined ? { required: f.required } : {}),
|
|
71
|
+
})),
|
|
72
|
+
},
|
|
73
|
+
...(schema.partitionSpec && schema.partitionSpec.length > 0
|
|
74
|
+
? {
|
|
75
|
+
partitionSpec: {
|
|
76
|
+
fields: schema.partitionSpec.map((p) => ({
|
|
77
|
+
name: p.name,
|
|
78
|
+
'source-id': p.sourceId,
|
|
79
|
+
transform: p.transform,
|
|
80
|
+
...(p.fieldId !== undefined ? { 'field-id': p.fieldId } : {}),
|
|
81
|
+
})),
|
|
82
|
+
},
|
|
83
|
+
}
|
|
84
|
+
: {}),
|
|
85
|
+
},
|
|
86
|
+
};
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* True when a `create*` failure should be swallowed as "the resource already
|
|
90
|
+
* exists". S3 Tables answers a duplicate name with `ConflictException`, whose
|
|
91
|
+
* only status in the service model is 409, and 409 is the signal this client has
|
|
92
|
+
* to match on: S3 Tables returns the exception name in an `x-amzn-ErrorType`
|
|
93
|
+
* header, while core's `parseError` (`packages/core/src/aws/signer.ts`) reads
|
|
94
|
+
* only the response body, and an S3 Tables error body carries `{"message": ...}`
|
|
95
|
+
* and nothing else. So every failure from this service arrives as
|
|
96
|
+
* `AwsError.code === "Http<status>"`, and `AwsError.isAlreadyExists` - which
|
|
97
|
+
* tests `code` against `/Conflict/i` - never matches here on its own. The
|
|
98
|
+
* `statusCode === 409` limb below is what actually makes `createTableBucket`,
|
|
99
|
+
* `createNamespace` and `createTable` idempotent, mirroring how `isNotFound`
|
|
100
|
+
* survives the same gap on its `statusCode === 404` limb.
|
|
101
|
+
*
|
|
102
|
+
* The accepted gap: S3 Tables documents `ConflictException` generically - "the
|
|
103
|
+
* request failed because there is a conflict with a previous write; retry" - and
|
|
104
|
+
* has no dedicated already-exists exception for these operations, so a genuine
|
|
105
|
+
* concurrent write conflict on `create*` reads as success here rather than being
|
|
106
|
+
* retried or surfaced. A confirming `get*` after the 409 would narrow it without
|
|
107
|
+
* any new signal, at the cost of a round trip on a path callers already reconcile
|
|
108
|
+
* read-then-create; not taking it is a deliberate trade, not a missing capability.
|
|
109
|
+
*
|
|
110
|
+
* The durable fix is core-level and deliberately not made here (this task must
|
|
111
|
+
* not touch `packages/core`): `parseError` should read `x-amzn-errortype` and
|
|
112
|
+
* `x-amzn-requestid` from the headers it already receives, which would hand every
|
|
113
|
+
* rest-json client its real error code and a request id to quote to AWS support,
|
|
114
|
+
* and would subsume this 409 limb. Until then, for this service
|
|
115
|
+
* `AwsError.requestId` is always `undefined` and any narrowing written as
|
|
116
|
+
* `err.code === 'NotFoundException'` silently never matches - worth knowing
|
|
117
|
+
* before adding one in tasks 34-36.
|
|
118
|
+
*/
|
|
119
|
+
function isAlreadyExists(err) {
|
|
120
|
+
return err instanceof AwsError && (err.isAlreadyExists || err.statusCode === 409);
|
|
121
|
+
}
|
|
122
|
+
/** S3 Tables control-plane client, over the shared SigV4 transport. */
|
|
123
|
+
export class S3TablesClient {
|
|
124
|
+
client;
|
|
125
|
+
constructor(client) {
|
|
126
|
+
this.client = client;
|
|
127
|
+
}
|
|
128
|
+
async call(method, path, payload, query) {
|
|
129
|
+
const res = await this.client.send({
|
|
130
|
+
service: SERVICE,
|
|
131
|
+
method,
|
|
132
|
+
path,
|
|
133
|
+
...(query ? { query } : {}),
|
|
134
|
+
headers: { 'content-type': 'application/json' },
|
|
135
|
+
...(payload !== undefined ? { body: JSON.stringify(payload) } : {}),
|
|
136
|
+
});
|
|
137
|
+
const text = res.text();
|
|
138
|
+
return (text ? JSON.parse(text) : {});
|
|
139
|
+
}
|
|
140
|
+
/**
|
|
141
|
+
* Create a table bucket. Idempotent: an already-existing bucket of the same
|
|
142
|
+
* name is not an error.
|
|
143
|
+
*
|
|
144
|
+
* Deliberately returns `void`, discarding the response's `arn`, rather than
|
|
145
|
+
* surfacing it: `getTableBucket` is ARN-keyed with no name-based lookup, so a
|
|
146
|
+
* caller that wants the ARN has to compute it before calling `createTableBucket`
|
|
147
|
+
* anyway - to run its own `getTableBucket` existence check first, per the usual
|
|
148
|
+
* read-then-create reconcile pattern - using the fixed
|
|
149
|
+
* `arn:aws:s3tables:<region>:<accountId>:bucket/<name>` form and the account id
|
|
150
|
+
* a `PluginContext` already carries. By the time `createTableBucket` runs, the
|
|
151
|
+
* caller already holds the ARN it needs; echoing the response's `arn` back would
|
|
152
|
+
* be redundant on the happy path, and unavailable on the already-exists path
|
|
153
|
+
* (the error body carries no `arn`), so returning it from only one of the two
|
|
154
|
+
* branches would be a false economy. `createNamespace` needs no such lookup at
|
|
155
|
+
* all (its identity is exactly its inputs); `createTable`'s identity is genuinely
|
|
156
|
+
* unrecoverable from its inputs (a table ARN carries an opaque generated id, not
|
|
157
|
+
* a name), but `getTable` is already name-keyed via `/get-table`'s query
|
|
158
|
+
* parameters, so a caller hydrates a table's ARN with a lookup, not by
|
|
159
|
+
* reconstructing it.
|
|
160
|
+
*/
|
|
161
|
+
async createTableBucket(name) {
|
|
162
|
+
try {
|
|
163
|
+
await this.call('PUT', PATHS.buckets, { name });
|
|
164
|
+
}
|
|
165
|
+
catch (err) {
|
|
166
|
+
if (isAlreadyExists(err))
|
|
167
|
+
return;
|
|
168
|
+
rethrowWithContext(err, 'createTableBucket', name);
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
/** Fetch a table bucket by ARN; undefined when it does not exist. */
|
|
172
|
+
async getTableBucket(tableBucketArn) {
|
|
173
|
+
try {
|
|
174
|
+
const res = await this.call('GET', PATHS.bucket(tableBucketArn));
|
|
175
|
+
return normalizeTableBucket(res);
|
|
176
|
+
}
|
|
177
|
+
catch (err) {
|
|
178
|
+
if (err instanceof AwsError && err.isNotFound)
|
|
179
|
+
return undefined;
|
|
180
|
+
rethrowWithContext(err, 'getTableBucket', tableBucketArn);
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
/** Delete a table bucket. No-op when it does not exist, so teardown is re-runnable. */
|
|
184
|
+
async deleteTableBucket(tableBucketArn) {
|
|
185
|
+
try {
|
|
186
|
+
await this.call('DELETE', PATHS.bucket(tableBucketArn));
|
|
187
|
+
}
|
|
188
|
+
catch (err) {
|
|
189
|
+
if (err instanceof AwsError && err.isNotFound)
|
|
190
|
+
return;
|
|
191
|
+
rethrowWithContext(err, 'deleteTableBucket', tableBucketArn);
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
/** Create a namespace in a table bucket. Idempotent: an already-existing namespace is not an error. */
|
|
195
|
+
async createNamespace(tableBucketArn, namespace) {
|
|
196
|
+
try {
|
|
197
|
+
await this.call('PUT', PATHS.namespaces(tableBucketArn), { namespace: [namespace] });
|
|
198
|
+
}
|
|
199
|
+
catch (err) {
|
|
200
|
+
if (isAlreadyExists(err))
|
|
201
|
+
return;
|
|
202
|
+
rethrowWithContext(err, 'createNamespace', `${tableBucketArn}/${namespace}`);
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
/** Fetch a namespace by bucket ARN and name; undefined when it does not exist. */
|
|
206
|
+
async getNamespace(tableBucketArn, namespace) {
|
|
207
|
+
try {
|
|
208
|
+
const res = await this.call('GET', PATHS.namespace(tableBucketArn, namespace));
|
|
209
|
+
return normalizeNamespace(res, tableBucketArn, namespace);
|
|
210
|
+
}
|
|
211
|
+
catch (err) {
|
|
212
|
+
if (err instanceof AwsError && err.isNotFound)
|
|
213
|
+
return undefined;
|
|
214
|
+
rethrowWithContext(err, 'getNamespace', `${tableBucketArn}/${namespace}`);
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
/** Delete a namespace. No-op when it does not exist, so teardown is re-runnable. */
|
|
218
|
+
async deleteNamespace(tableBucketArn, namespace) {
|
|
219
|
+
try {
|
|
220
|
+
await this.call('DELETE', PATHS.namespace(tableBucketArn, namespace));
|
|
221
|
+
}
|
|
222
|
+
catch (err) {
|
|
223
|
+
if (err instanceof AwsError && err.isNotFound)
|
|
224
|
+
return;
|
|
225
|
+
rethrowWithContext(err, 'deleteNamespace', `${tableBucketArn}/${namespace}`);
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
/**
|
|
229
|
+
* Create an Iceberg table in a namespace, carrying its schema (and, when given,
|
|
230
|
+
* its partition spec) so the table is never created schema-less. This matters
|
|
231
|
+
* beyond correctness-in-general: Firehose matches incoming record keys to
|
|
232
|
+
* Iceberg column names *exactly* and silently routes anything that does not
|
|
233
|
+
* match to the error bucket (see the analytics plugin spec's §Record
|
|
234
|
+
* transformation), so a schema-less table here fails every subsequent record
|
|
235
|
+
* with no error surfacing anywhere - the corruption this parameter exists to
|
|
236
|
+
* prevent. Idempotent: an already-existing table is not an error (its schema is
|
|
237
|
+
* not reconciled against `schema` on that path - S3 Tables has no
|
|
238
|
+
* update-schema-on-conflict operation for `CreateTable` to fall back to).
|
|
239
|
+
*/
|
|
240
|
+
async createTable(tableBucketArn, namespace, name, schema) {
|
|
241
|
+
try {
|
|
242
|
+
await this.call('PUT', PATHS.tables(tableBucketArn, namespace), {
|
|
243
|
+
name,
|
|
244
|
+
format: ICEBERG_FORMAT,
|
|
245
|
+
metadata: buildIcebergMetadata(schema),
|
|
246
|
+
});
|
|
247
|
+
}
|
|
248
|
+
catch (err) {
|
|
249
|
+
if (isAlreadyExists(err))
|
|
250
|
+
return;
|
|
251
|
+
rethrowWithContext(err, 'createTable', `${tableBucketArn}/${namespace}/${name}`);
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
/** Fetch a table by bucket ARN, namespace and name; undefined when it does not exist. */
|
|
255
|
+
async getTable(tableBucketArn, namespace, name) {
|
|
256
|
+
try {
|
|
257
|
+
const res = await this.call('GET', PATHS.getTable, undefined, {
|
|
258
|
+
tableBucketARN: tableBucketArn,
|
|
259
|
+
namespace,
|
|
260
|
+
name,
|
|
261
|
+
});
|
|
262
|
+
return normalizeTable(res, name);
|
|
263
|
+
}
|
|
264
|
+
catch (err) {
|
|
265
|
+
if (err instanceof AwsError && err.isNotFound)
|
|
266
|
+
return undefined;
|
|
267
|
+
rethrowWithContext(err, 'getTable', `${tableBucketArn}/${namespace}/${name}`);
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
/** Delete a table. No-op when it does not exist, so teardown is re-runnable. */
|
|
271
|
+
async deleteTable(tableBucketArn, namespace, name) {
|
|
272
|
+
try {
|
|
273
|
+
await this.call('DELETE', PATHS.table(tableBucketArn, namespace, name));
|
|
274
|
+
}
|
|
275
|
+
catch (err) {
|
|
276
|
+
if (err instanceof AwsError && err.isNotFound)
|
|
277
|
+
return;
|
|
278
|
+
rethrowWithContext(err, 'deleteTable', `${tableBucketArn}/${namespace}/${name}`);
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
}
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `blogwright analytics backfill`: the optional, one-shot, hand-run pull of
|
|
3
|
+
* history that predates the Firehose delivery, from the CloudWatch log group
|
|
4
|
+
* the site's own delivery already writes into the `page_views` table. This is
|
|
5
|
+
* the change spec's
|
|
6
|
+
* [§Analytics pipeline → Backfill of historical logs](../../../.specs/changes/merged/2026-07-26-analytics_plugin.md),
|
|
7
|
+
* and it is explicitly **not** part of the steady-state pipeline, which stays
|
|
8
|
+
* the push path §Shape draws: CloudFront → Firehose → Iceberg, with no
|
|
9
|
+
* operator in the loop.
|
|
10
|
+
*
|
|
11
|
+
* ## The identical-row property
|
|
12
|
+
*
|
|
13
|
+
* A record backfilled from CloudWatch and the same record delivered through
|
|
14
|
+
* Firehose produce the same `page_views` row, and that is a property of code
|
|
15
|
+
* reuse rather than of two implementations agreeing. This module owns no
|
|
16
|
+
* mapping at all: it hands each event to `mapRecord` - the same function the
|
|
17
|
+
* transform Lambda's envelope calls - which derives `event_time`, the `day`
|
|
18
|
+
* partition, `visitor_key` and `is_bot`, and applies the same drop rules. The
|
|
19
|
+
* historical day's salt is derivable for the same reason the spec gives: the
|
|
20
|
+
* per-day salt is `HMAC-SHA256(secret, day)` over one long-lived stored secret
|
|
21
|
+
* that is never rewritten, and `mapRecord` derives it from the record's own
|
|
22
|
+
* day, so a record from three months ago hashes under the salt it would have
|
|
23
|
+
* hashed under then.
|
|
24
|
+
*
|
|
25
|
+
* ## Idempotency, by construction rather than by de-duplication
|
|
26
|
+
*
|
|
27
|
+
* Three bounds, and none of them inspects a row to decide:
|
|
28
|
+
*
|
|
29
|
+
* 1. **Only whole UTC days strictly before the recorded bound.** The
|
|
30
|
+
* `analytics-log-delivery` node records {@link CREATED_DAY_KEY} the first
|
|
31
|
+
* time it creates its delivery, and never advances it. Firehose received
|
|
32
|
+
* nothing before its delivery existed, so every day this command touches is
|
|
33
|
+
* a day the Firehose path has no rows in, and the two row sets are
|
|
34
|
+
* disjoint. The boundary day itself is never backfilled - up to one day of
|
|
35
|
+
* history at the seam is the spec's stated precision limit, accepted rather
|
|
36
|
+
* than patched with a row-level de-duplication pass.
|
|
37
|
+
* 2. **A day that already holds rows is skipped**, counted through the
|
|
38
|
+
* existing `AnalyticsQuery` port. That is what makes a re-run a no-op and a
|
|
39
|
+
* crashed run resumable, and it is only sound because each day is inserted
|
|
40
|
+
* in one transaction: a partially written day would be counted as occupied
|
|
41
|
+
* and its remainder lost.
|
|
42
|
+
* 3. **A mapped row whose own `day` is not the day being written is not
|
|
43
|
+
* inserted.** The CloudWatch window is a request for a day's events and
|
|
44
|
+
* AWS's `endTime` is not documented as exclusive; the row's own `day` is,
|
|
45
|
+
* so that - not the window - is what decides which day a row belongs to.
|
|
46
|
+
* Without it a record on the far side of midnight could reach the day the
|
|
47
|
+
* Firehose delivery already covers.
|
|
48
|
+
*
|
|
49
|
+
* ## What it refuses, and why the refusal is not optional
|
|
50
|
+
*
|
|
51
|
+
* With no bound in the plugin's scoped state there is nothing to compute a
|
|
52
|
+
* range from, and there is no safe default: "everything" would insert days
|
|
53
|
+
* Firehose already delivered and silently double every row in them. So this
|
|
54
|
+
* command refuses, before any AWS call, in both of the two states that leave
|
|
55
|
+
* it without one - no delivery record at all, and a delivery record with no
|
|
56
|
+
* {@link CREATED_DAY_KEY}. The second is reachable and is not a corruption:
|
|
57
|
+
* the delivery node's `read` hydrates a delivery it finds already attached
|
|
58
|
+
* without writing the day, because `DescribeDeliveries` reports no creation
|
|
59
|
+
* date and a fabricated later bound is the one error direction that corrupts
|
|
60
|
+
* data rather than merely losing some.
|
|
61
|
+
*
|
|
62
|
+
* ## Ports
|
|
63
|
+
*
|
|
64
|
+
* The read is core's existing `LogsClient.filterEvents` over
|
|
65
|
+
* `ctx.clients.logsUsEast1` - no new client and no new core operation. The
|
|
66
|
+
* count is one named query through `AnalyticsQuery`. The write crosses
|
|
67
|
+
* `AnalyticsIngest`. This module names no vendor library and issues no
|
|
68
|
+
* statement of its own.
|
|
69
|
+
*/
|
|
70
|
+
import { type PluginContext } from 'blogwright-core';
|
|
71
|
+
import { type AnalyticsConfig } from './config.js';
|
|
72
|
+
import type { AnalyticsIngest, AnalyticsQuery } from './ports.js';
|
|
73
|
+
/**
|
|
74
|
+
* The slice of a plugin context this command reads, taken as a `Pick` of
|
|
75
|
+
* core's own `PluginContext` rather than a restatement of it, the way
|
|
76
|
+
* `DashboardCommandContext` and `DuckDbSessionContext` already are: the
|
|
77
|
+
* members cannot drift from the SPI, any `PluginContext<AnalyticsConfig>`
|
|
78
|
+
* satisfies it, and a test builds what it needs instead of the SPI's sixteen.
|
|
79
|
+
*
|
|
80
|
+
* `store`, `save`, `record` and `siteState` are deliberately absent, and their
|
|
81
|
+
* absence is a statement: **a backfill writes no state.** It reads the bound
|
|
82
|
+
* the delivery node recorded and nothing else, so a run that crashes halfway
|
|
83
|
+
* leaves the plugin's state object exactly as it found it, and the only thing
|
|
84
|
+
* that makes a second run different from the first is what is in the table.
|
|
85
|
+
*/
|
|
86
|
+
export type BackfillContext = Pick<PluginContext<AnalyticsConfig>, 'env' | 'config' | 'pluginConfig' | 'names' | 'clients' | 'state' | 'logger'>;
|
|
87
|
+
/** The two ports {@link runBackfill} is driven over. */
|
|
88
|
+
export interface BackfillPorts {
|
|
89
|
+
/** Reads the table - one named `row-count` per candidate day. */
|
|
90
|
+
readonly query: AnalyticsQuery;
|
|
91
|
+
/** Writes the table - one call per day that turned out to have rows. */
|
|
92
|
+
readonly ingest: AnalyticsIngest;
|
|
93
|
+
}
|
|
94
|
+
/**
|
|
95
|
+
* Run the backfill. Fails rather than continuing when a day's insert fails: a
|
|
96
|
+
* report saying five days landed while one in the middle did not would leave
|
|
97
|
+
* an operator unable to tell which history they have, and the occupancy check
|
|
98
|
+
* makes re-running after a fix cost nothing for the days that did land.
|
|
99
|
+
*/
|
|
100
|
+
export declare function runBackfill(ctx: BackfillContext, ports: BackfillPorts): Promise<void>;
|
package/dist/backfill.js
ADDED
|
@@ -0,0 +1,294 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `blogwright analytics backfill`: the optional, one-shot, hand-run pull of
|
|
3
|
+
* history that predates the Firehose delivery, from the CloudWatch log group
|
|
4
|
+
* the site's own delivery already writes into the `page_views` table. This is
|
|
5
|
+
* the change spec's
|
|
6
|
+
* [§Analytics pipeline → Backfill of historical logs](../../../.specs/changes/merged/2026-07-26-analytics_plugin.md),
|
|
7
|
+
* and it is explicitly **not** part of the steady-state pipeline, which stays
|
|
8
|
+
* the push path §Shape draws: CloudFront → Firehose → Iceberg, with no
|
|
9
|
+
* operator in the loop.
|
|
10
|
+
*
|
|
11
|
+
* ## The identical-row property
|
|
12
|
+
*
|
|
13
|
+
* A record backfilled from CloudWatch and the same record delivered through
|
|
14
|
+
* Firehose produce the same `page_views` row, and that is a property of code
|
|
15
|
+
* reuse rather than of two implementations agreeing. This module owns no
|
|
16
|
+
* mapping at all: it hands each event to `mapRecord` - the same function the
|
|
17
|
+
* transform Lambda's envelope calls - which derives `event_time`, the `day`
|
|
18
|
+
* partition, `visitor_key` and `is_bot`, and applies the same drop rules. The
|
|
19
|
+
* historical day's salt is derivable for the same reason the spec gives: the
|
|
20
|
+
* per-day salt is `HMAC-SHA256(secret, day)` over one long-lived stored secret
|
|
21
|
+
* that is never rewritten, and `mapRecord` derives it from the record's own
|
|
22
|
+
* day, so a record from three months ago hashes under the salt it would have
|
|
23
|
+
* hashed under then.
|
|
24
|
+
*
|
|
25
|
+
* ## Idempotency, by construction rather than by de-duplication
|
|
26
|
+
*
|
|
27
|
+
* Three bounds, and none of them inspects a row to decide:
|
|
28
|
+
*
|
|
29
|
+
* 1. **Only whole UTC days strictly before the recorded bound.** The
|
|
30
|
+
* `analytics-log-delivery` node records {@link CREATED_DAY_KEY} the first
|
|
31
|
+
* time it creates its delivery, and never advances it. Firehose received
|
|
32
|
+
* nothing before its delivery existed, so every day this command touches is
|
|
33
|
+
* a day the Firehose path has no rows in, and the two row sets are
|
|
34
|
+
* disjoint. The boundary day itself is never backfilled - up to one day of
|
|
35
|
+
* history at the seam is the spec's stated precision limit, accepted rather
|
|
36
|
+
* than patched with a row-level de-duplication pass.
|
|
37
|
+
* 2. **A day that already holds rows is skipped**, counted through the
|
|
38
|
+
* existing `AnalyticsQuery` port. That is what makes a re-run a no-op and a
|
|
39
|
+
* crashed run resumable, and it is only sound because each day is inserted
|
|
40
|
+
* in one transaction: a partially written day would be counted as occupied
|
|
41
|
+
* and its remainder lost.
|
|
42
|
+
* 3. **A mapped row whose own `day` is not the day being written is not
|
|
43
|
+
* inserted.** The CloudWatch window is a request for a day's events and
|
|
44
|
+
* AWS's `endTime` is not documented as exclusive; the row's own `day` is,
|
|
45
|
+
* so that - not the window - is what decides which day a row belongs to.
|
|
46
|
+
* Without it a record on the far side of midnight could reach the day the
|
|
47
|
+
* Firehose delivery already covers.
|
|
48
|
+
*
|
|
49
|
+
* ## What it refuses, and why the refusal is not optional
|
|
50
|
+
*
|
|
51
|
+
* With no bound in the plugin's scoped state there is nothing to compute a
|
|
52
|
+
* range from, and there is no safe default: "everything" would insert days
|
|
53
|
+
* Firehose already delivered and silently double every row in them. So this
|
|
54
|
+
* command refuses, before any AWS call, in both of the two states that leave
|
|
55
|
+
* it without one - no delivery record at all, and a delivery record with no
|
|
56
|
+
* {@link CREATED_DAY_KEY}. The second is reachable and is not a corruption:
|
|
57
|
+
* the delivery node's `read` hydrates a delivery it finds already attached
|
|
58
|
+
* without writing the day, because `DescribeDeliveries` reports no creation
|
|
59
|
+
* date and a fabricated later bound is the one error direction that corrupts
|
|
60
|
+
* data rather than merely losing some.
|
|
61
|
+
*
|
|
62
|
+
* ## Ports
|
|
63
|
+
*
|
|
64
|
+
* The read is core's existing `LogsClient.filterEvents` over
|
|
65
|
+
* `ctx.clients.logsUsEast1` - no new client and no new core operation. The
|
|
66
|
+
* count is one named query through `AnalyticsQuery`. The write crosses
|
|
67
|
+
* `AnalyticsIngest`. This module names no vendor library and issues no
|
|
68
|
+
* statement of its own.
|
|
69
|
+
*/
|
|
70
|
+
import { colors } from 'blogwright-core';
|
|
71
|
+
import { createAnalyticsClients } from './aws/clients.js';
|
|
72
|
+
import { resolveAnalyticsConfig } from './config.js';
|
|
73
|
+
import { CREATED_DAY_KEY, LOG_DELIVERY_NODE } from './nodes.js';
|
|
74
|
+
import { ROW_COUNT_COLUMN, ROW_COUNT_QUERY } from './queries.js';
|
|
75
|
+
import { mapRecord } from './transform/map-record.js';
|
|
76
|
+
/** Milliseconds in a UTC day. Every day boundary in this module is computed from it. */
|
|
77
|
+
const MS_PER_DAY = 86_400_000;
|
|
78
|
+
/** `YYYY-MM-DD` - the shape both the state bound and the `day` column carry. */
|
|
79
|
+
const DAY_PATTERN = /^\d{4}-\d{2}-\d{2}$/;
|
|
80
|
+
/** The leading characters of an ISO-8601 instant that are its UTC day. */
|
|
81
|
+
const ISO_DAY_LENGTH = 10;
|
|
82
|
+
/** The UTC day an epoch-millisecond instant falls on. */
|
|
83
|
+
function dayOf(milliseconds) {
|
|
84
|
+
return new Date(milliseconds).toISOString().slice(0, ISO_DAY_LENGTH);
|
|
85
|
+
}
|
|
86
|
+
/** Midnight UTC opening `day`, in epoch milliseconds. */
|
|
87
|
+
function startOf(day) {
|
|
88
|
+
return Date.parse(`${day}T00:00:00.000Z`);
|
|
89
|
+
}
|
|
90
|
+
/**
|
|
91
|
+
* True for a `YYYY-MM-DD` that names a day that exists. The round trip is what
|
|
92
|
+
* rejects `2026-02-30`: `Date.parse` accepts it and rolls it forward to March,
|
|
93
|
+
* so a shape check alone would compute a range from a day the calendar does
|
|
94
|
+
* not have.
|
|
95
|
+
*/
|
|
96
|
+
function isCalendarDay(day) {
|
|
97
|
+
if (!DAY_PATTERN.test(day))
|
|
98
|
+
return false;
|
|
99
|
+
const start = startOf(day);
|
|
100
|
+
return !Number.isNaN(start) && dayOf(start) === day;
|
|
101
|
+
}
|
|
102
|
+
/**
|
|
103
|
+
* The idempotency bound: the UTC day the plugin's delivery was first created,
|
|
104
|
+
* off the `analytics-log-delivery` node's recorded outputs.
|
|
105
|
+
*
|
|
106
|
+
* Three refusals rather than one, because the operator's remedy differs. No
|
|
107
|
+
* record at all means the pipeline was never provisioned. A record with no
|
|
108
|
+
* day means the state file lost the key or was written by a `read` that
|
|
109
|
+
* adopted an existing delivery - and there the remedy is to supply the bound,
|
|
110
|
+
* not to re-bootstrap, because bootstrapping again will not write a key the
|
|
111
|
+
* node only writes when it creates the delivery. A malformed day is neither,
|
|
112
|
+
* and says what it found.
|
|
113
|
+
*/
|
|
114
|
+
function requireCreatedDay(ctx) {
|
|
115
|
+
const bootstrap = `blogwright analytics bootstrap ${ctx.env}`;
|
|
116
|
+
const recorded = ctx.state.resources[LOG_DELIVERY_NODE];
|
|
117
|
+
if (recorded === undefined) {
|
|
118
|
+
throw new Error(`blogwright analytics backfill needs the day the Firehose delivery was created, and the analytics state for "${ctx.env}" records no ${LOG_DELIVERY_NODE} at all - run \`${bootstrap}\` to provision the pipeline first`);
|
|
119
|
+
}
|
|
120
|
+
const day = recorded[CREATED_DAY_KEY];
|
|
121
|
+
if (typeof day !== 'string' || day === '') {
|
|
122
|
+
throw new Error(`blogwright analytics backfill needs the day the Firehose delivery was created, and the ${LOG_DELIVERY_NODE} entry in the analytics state for "${ctx.env}" carries no "${CREATED_DAY_KEY}" - it is written only when the delivery is first created, so \`${bootstrap}\` will not add it to a delivery that already exists; set "${CREATED_DAY_KEY}" on that entry to the UTC day the delivery was created (a day too early loses nothing, a day too late double-inserts)`);
|
|
123
|
+
}
|
|
124
|
+
if (!isCalendarDay(day)) {
|
|
125
|
+
throw new Error(`the ${LOG_DELIVERY_NODE} entry in the analytics state for "${ctx.env}" carries "${CREATED_DAY_KEY}": ${JSON.stringify(day)}, which is not a YYYY-MM-DD calendar day - \`${bootstrap}\` writes it in that form`);
|
|
126
|
+
}
|
|
127
|
+
return day;
|
|
128
|
+
}
|
|
129
|
+
/**
|
|
130
|
+
* The candidate days, oldest first: the `retention.cloudfrontDays` whole UTC
|
|
131
|
+
* days immediately before `createdDay`.
|
|
132
|
+
*
|
|
133
|
+
* Anchored on the bound rather than on the clock, and that is deliberate -
|
|
134
|
+
* this command reads no clock at all. The lower bound exists because the log
|
|
135
|
+
* group is the only thing that holds these events and its retention is what
|
|
136
|
+
* decides how far back they go; anchoring it on "today" would need a clock
|
|
137
|
+
* and would make the same command answer differently on two consecutive days
|
|
138
|
+
* for no gain, since a day CloudWatch has already expired simply reads back
|
|
139
|
+
* with no events and is reported as such.
|
|
140
|
+
*/
|
|
141
|
+
function candidateDays(createdDay, retentionDays) {
|
|
142
|
+
const boundary = startOf(createdDay);
|
|
143
|
+
const days = [];
|
|
144
|
+
for (let back = retentionDays; back >= 1; back -= 1) {
|
|
145
|
+
days.push(dayOf(boundary - back * MS_PER_DAY));
|
|
146
|
+
}
|
|
147
|
+
return days;
|
|
148
|
+
}
|
|
149
|
+
/**
|
|
150
|
+
* The rows already in the table for `day`, through the named `row-count`
|
|
151
|
+
* query. `includeBots` is bound explicitly rather than left to
|
|
152
|
+
* `config.analytics.bots`: this is the table's occupancy and not a dashboard
|
|
153
|
+
* figure, so a day holding nothing but bot traffic is an occupied day.
|
|
154
|
+
*/
|
|
155
|
+
async function rowsAlreadyIn(query, day) {
|
|
156
|
+
const rows = await query.run(ROW_COUNT_QUERY, {
|
|
157
|
+
range: { from: day, to: day },
|
|
158
|
+
includeBots: true,
|
|
159
|
+
});
|
|
160
|
+
const count = rows[0]?.[ROW_COUNT_COLUMN];
|
|
161
|
+
if (typeof count !== 'number') {
|
|
162
|
+
throw new Error(`the ${ROW_COUNT_QUERY} query answered no ${ROW_COUNT_COLUMN} for ${day}, so the backfill cannot tell whether that day is already in the table`);
|
|
163
|
+
}
|
|
164
|
+
return count;
|
|
165
|
+
}
|
|
166
|
+
/**
|
|
167
|
+
* One CloudWatch log event's message as a CloudFront record, or `undefined`
|
|
168
|
+
* when it is not one.
|
|
169
|
+
*
|
|
170
|
+
* The twin of `transform/handler.ts`'s `decodePayload`, and separate from it
|
|
171
|
+
* on purpose: the two envelopes differ (Firehose hands over base64, CloudWatch
|
|
172
|
+
* hands over the message text), and `map-record.ts` documents that parsing and
|
|
173
|
+
* its failures belong to each boundary rather than to the mapping. What the
|
|
174
|
+
* two share - what a record IS once parsed - is the type, not the parser.
|
|
175
|
+
*/
|
|
176
|
+
function recordFrom(message) {
|
|
177
|
+
let parsed;
|
|
178
|
+
try {
|
|
179
|
+
parsed = JSON.parse(message);
|
|
180
|
+
}
|
|
181
|
+
catch {
|
|
182
|
+
return undefined;
|
|
183
|
+
}
|
|
184
|
+
if (parsed === null || typeof parsed !== 'object' || Array.isArray(parsed))
|
|
185
|
+
return undefined;
|
|
186
|
+
return parsed;
|
|
187
|
+
}
|
|
188
|
+
/**
|
|
189
|
+
* Map one day's log events. Every event goes through `mapRecord` and nothing
|
|
190
|
+
* else; what this adds is the two ways a mapped event still does not belong in
|
|
191
|
+
* this day's insert - it could not be mapped at all, or its own `day` says it
|
|
192
|
+
* belongs to another one.
|
|
193
|
+
*/
|
|
194
|
+
function mapDay(messages, day, saltSecret) {
|
|
195
|
+
const rows = [];
|
|
196
|
+
let unmappable = 0;
|
|
197
|
+
let foreign = 0;
|
|
198
|
+
let firstReason;
|
|
199
|
+
for (const message of messages) {
|
|
200
|
+
const record = recordFrom(message);
|
|
201
|
+
if (record === undefined) {
|
|
202
|
+
unmappable += 1;
|
|
203
|
+
firstReason ??= 'the log event is not a JSON object';
|
|
204
|
+
continue;
|
|
205
|
+
}
|
|
206
|
+
const mapped = mapRecord(record, saltSecret);
|
|
207
|
+
if (!mapped.mapped) {
|
|
208
|
+
unmappable += 1;
|
|
209
|
+
firstReason ??= mapped.reason;
|
|
210
|
+
continue;
|
|
211
|
+
}
|
|
212
|
+
if (mapped.row.day !== day) {
|
|
213
|
+
foreign += 1;
|
|
214
|
+
continue;
|
|
215
|
+
}
|
|
216
|
+
rows.push(mapped.row);
|
|
217
|
+
}
|
|
218
|
+
return { rows, unmappable, foreign, ...(firstReason === undefined ? {} : { firstReason }) };
|
|
219
|
+
}
|
|
220
|
+
/**
|
|
221
|
+
* The long-lived salt secret behind `visitor_key`, read once for the whole
|
|
222
|
+
* run through the plugin's own us-east-1 Secrets Manager client - the region
|
|
223
|
+
* pin, so the value read here is the value the transform Lambda reads.
|
|
224
|
+
*
|
|
225
|
+
* Read before the first day rather than on the first day that needs it: a run
|
|
226
|
+
* that discovered a missing secret after eighty-nine occupancy queries would
|
|
227
|
+
* have spent them for nothing, and an operator who has to fix the secret wants
|
|
228
|
+
* to know at the start.
|
|
229
|
+
*/
|
|
230
|
+
async function readSaltSecret(ctx, secretName) {
|
|
231
|
+
const secret = await createAnalyticsClients(ctx).secrets.getSecretValue(secretName);
|
|
232
|
+
if (secret === undefined || secret.trim() === '') {
|
|
233
|
+
throw new Error(`the analytics salt secret "${secretName}" holds no value: an unsalted visitor_key would identify the visitor it exists to hide, so the backfill stops instead - run \`blogwright analytics bootstrap ${ctx.env}\` to create the secret`);
|
|
234
|
+
}
|
|
235
|
+
return secret;
|
|
236
|
+
}
|
|
237
|
+
/** One day: count, read, map, insert - or say why it was skipped. */
|
|
238
|
+
async function backfillDay(ctx, ports, day, saltSecret) {
|
|
239
|
+
const occupied = await rowsAlreadyIn(ports.query, day);
|
|
240
|
+
if (occupied > 0) {
|
|
241
|
+
ctx.logger.info(` skipped ${day}: the table already holds ${occupied} rows for that day`);
|
|
242
|
+
return { day, inserted: 0, skipped: 'occupied' };
|
|
243
|
+
}
|
|
244
|
+
const start = startOf(day);
|
|
245
|
+
const events = await ctx.clients.logsUsEast1.filterEvents(ctx.names.cloudfrontLogGroup, {
|
|
246
|
+
startTime: start,
|
|
247
|
+
endTime: start + MS_PER_DAY,
|
|
248
|
+
});
|
|
249
|
+
const mapped = mapDay(events.map((event) => event.message), day, saltSecret);
|
|
250
|
+
// Reported per day rather than aggregated, and reported at all because the
|
|
251
|
+
// silent version of this is the failure mode the whole pipeline is written
|
|
252
|
+
// against: an operator whose log group carries a different field set sees an
|
|
253
|
+
// empty table and no error anywhere.
|
|
254
|
+
if (mapped.unmappable > 0) {
|
|
255
|
+
ctx.logger.warn(`${day}: ${mapped.unmappable} of ${events.length} log events could not be mapped and were not inserted - ${mapped.firstReason}`);
|
|
256
|
+
}
|
|
257
|
+
if (mapped.foreign > 0) {
|
|
258
|
+
ctx.logger.warn(`${day}: ${mapped.foreign} log events carried another day and were left to that day's own pass`);
|
|
259
|
+
}
|
|
260
|
+
if (mapped.rows.length === 0) {
|
|
261
|
+
ctx.logger.info(` skipped ${day}: the log group holds no events for that day`);
|
|
262
|
+
return { day, inserted: 0, skipped: 'empty' };
|
|
263
|
+
}
|
|
264
|
+
await ports.ingest.insertDay(day, mapped.rows);
|
|
265
|
+
ctx.logger.info(` inserted ${day}: ${mapped.rows.length} rows`);
|
|
266
|
+
return { day, inserted: mapped.rows.length };
|
|
267
|
+
}
|
|
268
|
+
/** Count the outcomes carrying `skipped`. */
|
|
269
|
+
function countSkipped(outcomes, reason) {
|
|
270
|
+
return outcomes.filter((outcome) => outcome.skipped === reason).length;
|
|
271
|
+
}
|
|
272
|
+
/**
|
|
273
|
+
* Run the backfill. Fails rather than continuing when a day's insert fails: a
|
|
274
|
+
* report saying five days landed while one in the middle did not would leave
|
|
275
|
+
* an operator unable to tell which history they have, and the occupancy check
|
|
276
|
+
* makes re-running after a fix cost nothing for the days that did land.
|
|
277
|
+
*/
|
|
278
|
+
export async function runBackfill(ctx, ports) {
|
|
279
|
+
const createdDay = requireCreatedDay(ctx);
|
|
280
|
+
const config = resolveAnalyticsConfig(ctx);
|
|
281
|
+
const relation = `${config.namespace}.${config.table}`;
|
|
282
|
+
const days = candidateDays(createdDay, ctx.config.retention.cloudfrontDays);
|
|
283
|
+
ctx.logger.info(colors.bold(`Analytics backfill for "${ctx.env}" from ${ctx.names.cloudfrontLogGroup} into ${relation}`));
|
|
284
|
+
ctx.logger.info(` ${days.length} whole UTC days from ${days[0]} to ${days.at(-1)}, bounded below by retention.cloudfrontDays`);
|
|
285
|
+
const saltSecret = await readSaltSecret(ctx, config.saltSecretName);
|
|
286
|
+
const outcomes = [];
|
|
287
|
+
for (const day of days) {
|
|
288
|
+
outcomes.push(await backfillDay(ctx, ports, day, saltSecret));
|
|
289
|
+
}
|
|
290
|
+
const insertedDays = outcomes.filter((outcome) => outcome.inserted > 0);
|
|
291
|
+
const insertedRows = insertedDays.reduce((total, outcome) => total + outcome.inserted, 0);
|
|
292
|
+
ctx.logger.ok(`backfill complete: inserted ${insertedRows} rows across ${insertedDays.length} days; skipped ${countSkipped(outcomes, 'occupied')} days already in the table and ${countSkipped(outcomes, 'empty')} with no events`);
|
|
293
|
+
ctx.logger.info(` ${createdDay} is the day the Firehose delivery was created and is never backfilled - up to one day of history at the seam is the accepted precision limit`);
|
|
294
|
+
}
|