blogwright-analytics 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +162 -0
- package/dist/adapters/duckdb-ingest.d.ts +76 -0
- package/dist/adapters/duckdb-ingest.js +173 -0
- package/dist/adapters/duckdb-query.d.ts +56 -0
- package/dist/adapters/duckdb-query.js +80 -0
- package/dist/adapters/duckdb-session.d.ts +168 -0
- package/dist/adapters/duckdb-session.js +330 -0
- package/dist/app/_app/immutable/assets/0.BTQrrh5B.css +1 -0
- package/dist/app/_app/immutable/assets/2.CZSK3rT8.css +1 -0
- package/dist/app/_app/immutable/assets/BrushContext.D7c8UPey.css +1 -0
- package/dist/app/_app/immutable/assets/ChartAnnotations.CPxIG7Mw.css +1 -0
- package/dist/app/_app/immutable/assets/Circle.C5MKzgk2.css +1 -0
- package/dist/app/_app/immutable/assets/DefaultTooltip.C5-uctZ7.css +1 -0
- package/dist/app/_app/immutable/assets/Group.DV48xipa.css +1 -0
- package/dist/app/_app/immutable/assets/Labels.BxZ4NUVz.css +1 -0
- package/dist/app/_app/immutable/assets/Legend.CxnrE4Ye.css +1 -0
- package/dist/app/_app/immutable/assets/Line.fkmsECm9.css +1 -0
- package/dist/app/_app/immutable/assets/Path.CvpwNZ6g.css +1 -0
- package/dist/app/_app/immutable/assets/Rect.CtRaGMmQ.css +1 -0
- package/dist/app/_app/immutable/assets/Text.j9l35qB0.css +1 -0
- package/dist/app/_app/immutable/assets/TransformContext.Bs_HkpAk.css +1 -0
- package/dist/app/_app/immutable/assets/Voronoi.ce7atosu.css +1 -0
- package/dist/app/_app/immutable/chunks/-aNGNaBT.js +1 -0
- package/dist/app/_app/immutable/chunks/6djn-yLs.js +1 -0
- package/dist/app/_app/immutable/chunks/B1amyutE.js +1 -0
- package/dist/app/_app/immutable/chunks/B3vZDoek.js +1 -0
- package/dist/app/_app/immutable/chunks/B5KRA4hC.js +1 -0
- package/dist/app/_app/immutable/chunks/BClnVG6H.js +1 -0
- package/dist/app/_app/immutable/chunks/BID1NNRh.js +1 -0
- package/dist/app/_app/immutable/chunks/BR2LaRms.js +1 -0
- package/dist/app/_app/immutable/chunks/Bd1gDe3Y.js +1 -0
- package/dist/app/_app/immutable/chunks/Bjy-W4x2.js +81 -0
- package/dist/app/_app/immutable/chunks/Bl052uUt.js +1 -0
- package/dist/app/_app/immutable/chunks/Bye3lL0c.js +1 -0
- package/dist/app/_app/immutable/chunks/C58PZtCD.js +4 -0
- package/dist/app/_app/immutable/chunks/CAzydqEO.js +1 -0
- package/dist/app/_app/immutable/chunks/CCch3uox.js +1 -0
- package/dist/app/_app/immutable/chunks/CIlSMUH9.js +1 -0
- package/dist/app/_app/immutable/chunks/CO1vUXfR.js +1 -0
- package/dist/app/_app/immutable/chunks/CPbD8C65.js +5 -0
- package/dist/app/_app/immutable/chunks/CRTcXoMo.js +1 -0
- package/dist/app/_app/immutable/chunks/CjjyIQAO.js +1 -0
- package/dist/app/_app/immutable/chunks/CuXAxjvF.js +1 -0
- package/dist/app/_app/immutable/chunks/CvyVA_jC.js +1 -0
- package/dist/app/_app/immutable/chunks/CxGCFVdy.js +1 -0
- package/dist/app/_app/immutable/chunks/D0Ty6LN0.js +1 -0
- package/dist/app/_app/immutable/chunks/D2AaQUUW.js +1 -0
- package/dist/app/_app/immutable/chunks/D2BnX0Uk.js +3 -0
- package/dist/app/_app/immutable/chunks/DJc8C0NK.js +1 -0
- package/dist/app/_app/immutable/chunks/DKMlMI4a.js +1 -0
- package/dist/app/_app/immutable/chunks/DVXZkpbf.js +1 -0
- package/dist/app/_app/immutable/chunks/DVt8ukQ_.js +1 -0
- package/dist/app/_app/immutable/chunks/DZPlYdq_.js +1 -0
- package/dist/app/_app/immutable/chunks/Db0q5_zr.js +1 -0
- package/dist/app/_app/immutable/chunks/Dfvzj6n2.js +1 -0
- package/dist/app/_app/immutable/chunks/Dh958be7.js +1 -0
- package/dist/app/_app/immutable/chunks/DjKLLdnY.js +15 -0
- package/dist/app/_app/immutable/chunks/Doz7YX1W.js +1 -0
- package/dist/app/_app/immutable/chunks/DthYhn6Y.js +2 -0
- package/dist/app/_app/immutable/chunks/DtuTIrAM.js +1 -0
- package/dist/app/_app/immutable/chunks/HclGiUj8.js +1 -0
- package/dist/app/_app/immutable/chunks/Hx0TNsV3.js +1 -0
- package/dist/app/_app/immutable/chunks/RobXhXPM.js +1 -0
- package/dist/app/_app/immutable/chunks/V9ZjaxiY.js +1 -0
- package/dist/app/_app/immutable/chunks/Y5urAfNy.js +1 -0
- package/dist/app/_app/immutable/chunks/caXkbKD3.js +1 -0
- package/dist/app/_app/immutable/chunks/devYm2ud.js +1 -0
- package/dist/app/_app/immutable/chunks/mtZWP0zR.js +1 -0
- package/dist/app/_app/immutable/chunks/vDgBJUjM.js +1 -0
- package/dist/app/_app/immutable/chunks/xIq_fFFM.js +1 -0
- package/dist/app/_app/immutable/chunks/xihTtKlq.js +1 -0
- package/dist/app/_app/immutable/chunks/z05MoCFz.js +1 -0
- package/dist/app/_app/immutable/entry/app.CLAerUAN.js +2 -0
- package/dist/app/_app/immutable/entry/start.D3MqnNci.js +1 -0
- package/dist/app/_app/immutable/nodes/0.UTMEigHJ.js +1 -0
- package/dist/app/_app/immutable/nodes/1.Cn4f11bT.js +1 -0
- package/dist/app/_app/immutable/nodes/2.B39cIcr2.js +6 -0
- package/dist/app/_app/version.json +1 -0
- package/dist/app/index.html +82 -0
- package/dist/aws/clients.d.ts +70 -0
- package/dist/aws/clients.js +52 -0
- package/dist/aws/errors.d.ts +41 -0
- package/dist/aws/errors.js +70 -0
- package/dist/aws/firehose.d.ts +228 -0
- package/dist/aws/firehose.js +347 -0
- package/dist/aws/glue.d.ts +103 -0
- package/dist/aws/glue.js +225 -0
- package/dist/aws/lambda.d.ts +132 -0
- package/dist/aws/lambda.js +339 -0
- package/dist/aws/s3tables.d.ts +120 -0
- package/dist/aws/s3tables.js +281 -0
- package/dist/backfill.d.ts +100 -0
- package/dist/backfill.js +294 -0
- package/dist/commands.d.ts +124 -0
- package/dist/commands.js +336 -0
- package/dist/config.d.ts +162 -0
- package/dist/config.js +317 -0
- package/dist/fixture-ingest.d.ts +49 -0
- package/dist/fixture-ingest.js +43 -0
- package/dist/fixture-query.d.ts +39 -0
- package/dist/fixture-query.js +70 -0
- package/dist/index.d.ts +35 -0
- package/dist/index.js +35 -0
- package/dist/nodes.d.ts +404 -0
- package/dist/nodes.js +2708 -0
- package/dist/paths.d.ts +45 -0
- package/dist/paths.js +47 -0
- package/dist/plugin.d.ts +102 -0
- package/dist/plugin.js +248 -0
- package/dist/ports.d.ts +113 -0
- package/dist/ports.js +35 -0
- package/dist/queries.d.ts +301 -0
- package/dist/queries.js +414 -0
- package/dist/schema.d.ts +240 -0
- package/dist/schema.js +154 -0
- package/dist/server.d.ts +150 -0
- package/dist/server.js +499 -0
- package/dist/transform/bots.d.ts +47 -0
- package/dist/transform/bots.js +73 -0
- package/dist/transform/handler.d.ts +135 -0
- package/dist/transform/handler.js +177 -0
- package/dist/transform/map-record.d.ts +110 -0
- package/dist/transform/map-record.js +275 -0
- package/dist/transform/visitor-key.d.ts +83 -0
- package/dist/transform/visitor-key.js +120 -0
- package/dist/transform-bundle/index.mjs +21456 -0
- package/dist/transform-bundle/transform-manifest.json +4 -0
- package/dist/transform-hash.d.ts +135 -0
- package/dist/transform-hash.js +186 -0
- package/dist/write-transform-manifest.mjs +365 -0
- package/package.json +59 -0
package/README.md
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
# blogwright-analytics
|
|
2
|
+
|
|
3
|
+
Traffic analytics for [blogwright](https://github.com/antstanley/blogwright): a second
|
|
4
|
+
CloudFront access-log delivery routed through Amazon Data Firehose into an Apache Iceberg
|
|
5
|
+
table in an S3 Tables bucket, with a local SvelteKit dashboard that reads that table
|
|
6
|
+
through DuckDB. The site's existing CloudWatch log delivery is untouched - one CloudFront
|
|
7
|
+
delivery source carries several deliveries, so this one is added beside it.
|
|
8
|
+
|
|
9
|
+
This package owns the whole pipeline: its own four AWS service clients (S3 Tables,
|
|
10
|
+
Firehose, Glue, Lambda), the twelve resource nodes those clients reconcile, the
|
|
11
|
+
record-transform Lambda, the named query set, and the dashboard application. It depends
|
|
12
|
+
on `blogwright-core` (ports, the SigV4 transport and signer, the S3 and Secrets Manager
|
|
13
|
+
clients) and on DuckDB, which is reached only through the `AnalyticsQuery` port's one
|
|
14
|
+
adapter. It never imports the CLI.
|
|
15
|
+
|
|
16
|
+
## Install
|
|
17
|
+
|
|
18
|
+
The plugin is not shipped with the CLI. A repo that never installs it pays nothing:
|
|
19
|
+
|
|
20
|
+
```sh
|
|
21
|
+
blogwright plugin add analytics
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
`analytics` resolves to the package `blogwright-analytics`, installed at the running
|
|
25
|
+
CLI's own version and pinned exactly, so the CLI and its plugins cannot drift apart
|
|
26
|
+
between two checkouts of the same repo. `blogwright plugin list` shows it once installed,
|
|
27
|
+
and `blogwright plugin remove analytics` offers to tear its resources down first.
|
|
28
|
+
|
|
29
|
+
## Commands
|
|
30
|
+
|
|
31
|
+
Once installed, the plugin answers `blogwright analytics <action> [env]`. The environment
|
|
32
|
+
defaults to `production` and `--env` overrides the positional, exactly as for a built-in
|
|
33
|
+
command. Five actions are the steady state:
|
|
34
|
+
|
|
35
|
+
| Action | What it does |
|
|
36
|
+
| --- | --- |
|
|
37
|
+
| `blogwright analytics init [env]` | Asks for the `analytics` config block - namespace, table, bot handling, dashboard port - and splices it into `config/<env>.jsonc`. `blogwright init` asks the same questions as part of the first-run wizard. |
|
|
38
|
+
| `blogwright analytics bootstrap [env]` | Provisions the pipeline: the S3 Tables bucket, its namespace and the `page_views` table, the Glue `s3tablescatalog` federation, the `visitor_key` salt secret, the transform Lambda and its execution role, the Firehose error bucket, delivery role and delivery stream, and the CloudWatch delivery destination and delivery. |
|
|
39
|
+
| `blogwright analytics status [env]` | Reports each node present or missing, then the Firehose stream's delivery health and the table's current row count. An environment that was never bootstrapped is not an error: every node reports missing and the command exits 0. |
|
|
40
|
+
| `blogwright analytics dashboard [env]` | Serves the prebuilt dashboard from `dist/app` on `127.0.0.1` (port 4317 by default) and answers its named queries against the table. |
|
|
41
|
+
| `blogwright analytics destroy [env] --yes` | Removes the plugin's own resources. The Glue federation is the one exception and is never deleted - see below. |
|
|
42
|
+
|
|
43
|
+
Two things the table does not say. The plugin's resources are recorded in their own state
|
|
44
|
+
object, `state/<env>.analytics.json`: `blogwright bootstrap` provisions none of them, and
|
|
45
|
+
`blogwright destroy --yes` refuses while that object exists and names
|
|
46
|
+
`blogwright analytics destroy <env> --yes` as the way through. And the dashboard's server
|
|
47
|
+
answers only the queries this package defines by name - the seven the dashboard charts
|
|
48
|
+
(`views-over-time`, `unique-visitors`, `top-paths`, `referrers`, `countries`,
|
|
49
|
+
`status-codes`, `cache-hit-ratio`) plus the `row-count` that `analytics status` reads.
|
|
50
|
+
It never executes SQL supplied over its socket, and it binds loopback only.
|
|
51
|
+
|
|
52
|
+
### `blogwright analytics backfill [env]` - optional, one-shot
|
|
53
|
+
|
|
54
|
+
A sixth action, deliberately not in the table above, because it is not part of
|
|
55
|
+
the steady state. Firehose only carries what CloudFront produced after its
|
|
56
|
+
delivery existed; `backfill` is the hand-run pull of the history that came
|
|
57
|
+
before it, and once it has run there is no reason to run it again.
|
|
58
|
+
|
|
59
|
+
It reads the CloudWatch log group the site's own delivery already writes -
|
|
60
|
+
`/<siteName>/<env>/cloudfront`, bounded by `retention.cloudfrontDays` - and
|
|
61
|
+
maps every event through the same code the transform Lambda runs, so a record
|
|
62
|
+
produces the same `page_views` row whichever path carried it, `visitor_key`
|
|
63
|
+
included: the day's salt is `HMAC-SHA256(secret, day)` over the same stored
|
|
64
|
+
secret, so a historical day's salt is derivable and the raw IP is no more
|
|
65
|
+
stored here than it is there.
|
|
66
|
+
|
|
67
|
+
**It cannot double-count, and not by de-duplicating.** The
|
|
68
|
+
`analytics-log-delivery` node records the UTC day it first created the
|
|
69
|
+
delivery, once and never again, and the backfill inserts only whole days
|
|
70
|
+
*strictly before* that day - Firehose received nothing before its delivery
|
|
71
|
+
existed, so the two paths never write the same day. Within that range each day
|
|
72
|
+
is one transaction, a day the table already holds rows for is skipped, and a
|
|
73
|
+
row whose own `day` is not the day being written is not inserted. So a re-run
|
|
74
|
+
inserts nothing, and a run that crashed resumes where it stopped.
|
|
75
|
+
|
|
76
|
+
The boundary day itself is never backfilled. Up to one day of history at the
|
|
77
|
+
seam is lost, which is the accepted precision limit rather than an oversight:
|
|
78
|
+
buying it back would mean comparing rows, and comparing rows is the thing this
|
|
79
|
+
design does not do.
|
|
80
|
+
|
|
81
|
+
The command refuses, before it calls AWS at all, when the plugin's state
|
|
82
|
+
carries no delivery record - run `blogwright analytics bootstrap <env>` first -
|
|
83
|
+
and also when it carries a delivery with no recorded day, which is what a state
|
|
84
|
+
file that lost the key looks like. There is no default in that case: assuming
|
|
85
|
+
"everything" would insert days Firehose has already delivered and double every
|
|
86
|
+
row in them, so the command says what to supply instead. Its report names every
|
|
87
|
+
day it inserted, every day it skipped and why, and the boundary day it left
|
|
88
|
+
alone.
|
|
89
|
+
|
|
90
|
+
## Everything is created in us-east-1
|
|
91
|
+
|
|
92
|
+
Every resource this plugin owns is created in `us-east-1` regardless of `config.region`,
|
|
93
|
+
and every node's title says so rather than diverging silently.
|
|
94
|
+
|
|
95
|
+
CloudFront forces it. CloudFront is a global service whose logging control plane lives in
|
|
96
|
+
`us-east-1` alone, and standard logging accepts a Firehose delivery stream only there, so
|
|
97
|
+
the stream has to be in that region. Everything the stream reaches has to follow: the
|
|
98
|
+
table bucket and its Iceberg table, the Glue federation the stream writes through, the
|
|
99
|
+
transform Lambda the stream invokes, the Firehose error bucket, and the salt secret the
|
|
100
|
+
Lambda reads. The two IAM roles are global, and state the pipeline they serve instead.
|
|
101
|
+
|
|
102
|
+
That is why this package builds its own S3 and Secrets Manager clients over the host's
|
|
103
|
+
`signingUsEast1` signer rather than reusing `ctx.clients.s3` and `ctx.clients.secrets`:
|
|
104
|
+
the host's pair signs in `config.region`, which would put the error bucket, and the salt,
|
|
105
|
+
in a region the transform function cannot read from.
|
|
106
|
+
|
|
107
|
+
## Privacy
|
|
108
|
+
|
|
109
|
+
**The raw viewer IP is never stored.** `c-ip` is selected from CloudFront only so the
|
|
110
|
+
transform Lambda can derive `visitor_key` from it, and no column of the `page_views` table
|
|
111
|
+
holds it: it has no entry in the field-to-column map, and the transform discards it after
|
|
112
|
+
hashing. `visitor_key` is a SHA-256 digest over the viewer IP, the user agent and that
|
|
113
|
+
day's salt, and the salt is `HMAC-SHA256(secret, day)` over one long-lived random secret
|
|
114
|
+
held in Secrets Manager - never the date alone, which anyone holding the table could
|
|
115
|
+
compute and then brute-force back across a 32-bit address space. The consequence is
|
|
116
|
+
deliberate and stated where it matters: a `visitor_key` is not comparable across days, so
|
|
117
|
+
a monthly unique-visitor figure is the sum of daily uniques rather than a distinct count.
|
|
118
|
+
|
|
119
|
+
**`cs(Cookie)` and `x-forwarded-for` are never selected**, so they never leave CloudFront
|
|
120
|
+
for this pipeline: they reach neither Firehose, nor the transform, nor the table. This
|
|
121
|
+
governs the analytics delivery only. The site's existing CloudWatch delivery is created
|
|
122
|
+
with no field list, so AWS's default set - which includes both - still applies to that
|
|
123
|
+
copy; narrowing it is a change to the site's own node, not this one.
|
|
124
|
+
|
|
125
|
+
No cookie is set and no identifier is written to a visitor's browser. Bot traffic is
|
|
126
|
+
flagged rather than dropped (`is_bot` is a column, and filtering is a query default), so
|
|
127
|
+
a heuristic that turns out to be wrong is a query change and not lost data.
|
|
128
|
+
|
|
129
|
+
## Shared state, and what teardown leaves behind
|
|
130
|
+
|
|
131
|
+
One resource is account-and-region scoped rather than per-environment: the Glue
|
|
132
|
+
`s3tablescatalog` federation that Firehose reaches S3 Tables through. Two environments of
|
|
133
|
+
the same site share it. Its node therefore adopts an existing federation rather than
|
|
134
|
+
failing, and its `delete()` is a no-op - tearing down staging must not break production,
|
|
135
|
+
and the Glue API this package speaks exposes no delete operation at all. Everything else
|
|
136
|
+
the plugin creates is per-environment and is removed by
|
|
137
|
+
`blogwright analytics destroy <env> --yes`.
|
|
138
|
+
|
|
139
|
+
Rows are never aged out. The table is append-only and partitioned by `day`, so expiring
|
|
140
|
+
old data would mean whole-partition deletes issued on a schedule; S3 Tables offers no
|
|
141
|
+
row-retention setting for a table you create. The site's `retention.cloudfrontDays`
|
|
142
|
+
governs only the CloudWatch copy of the logs.
|
|
143
|
+
|
|
144
|
+
## Configuration
|
|
145
|
+
|
|
146
|
+
The `analytics` block in `config/<env>.jsonc`, all of it optional:
|
|
147
|
+
|
|
148
|
+
| Key | Default | Meaning |
|
|
149
|
+
| --- | --- | --- |
|
|
150
|
+
| `namespace` | `web` | Iceberg namespace holding the table. |
|
|
151
|
+
| `table` | `page_views` | Iceberg table the page views land in. |
|
|
152
|
+
| `bots` | `flag` | `flag` keeps bot rows and marks them; `filter` excludes them from queries. |
|
|
153
|
+
| `dashboard.port` | `4317` | Port the local dashboard binds on `127.0.0.1`. |
|
|
154
|
+
| `tableBucket` | `<env>-<siteName>-analytics` | S3 Tables bucket holding the namespace. |
|
|
155
|
+
| `saltSecretName` | `<siteName>/<env>/analytics-salt` | Secrets Manager secret the daily salt is derived from. |
|
|
156
|
+
|
|
157
|
+
The last two carry the environment in their defaults and are not asked by the wizard: a
|
|
158
|
+
prompt whose default is wrong for every environment but one is worse than no prompt. Write
|
|
159
|
+
them into the block by hand to override either, where the same validator still checks
|
|
160
|
+
them. Do not take the environment out of either default: without it two environments
|
|
161
|
+
resolve to the same Iceberg table and the same salt, and `blogwright analytics destroy`
|
|
162
|
+
in staging would delete production's data.
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The real {@link AnalyticsIngest}: DuckDB with the S3 Tables catalog attached
|
|
3
|
+
* writable, inserting one whole UTC day per transaction. It is the write half
|
|
4
|
+
* of the change spec's §Backfill of historical logs - "written through the
|
|
5
|
+
* DuckDB dependency the dashboard already ships, behind a write port of its
|
|
6
|
+
* own" - and it shares every part of the session with the read adapter beside
|
|
7
|
+
* it through `duckdb-session.ts`: the same credential secret, the same attach
|
|
8
|
+
* target, the same quoting and the same rule that no vendor error object
|
|
9
|
+
* escapes. The one clause that differs is `READ_ONLY`, which this session
|
|
10
|
+
* omits.
|
|
11
|
+
*
|
|
12
|
+
* **The steady-state pipeline does not come through here.** Firehose writes
|
|
13
|
+
* the table itself; this module exists for the one-shot `analytics backfill`
|
|
14
|
+
* action and is constructed only by that command. The dashboard's session is a
|
|
15
|
+
* different object with `readOnly: true`, so nothing the server can be asked
|
|
16
|
+
* to do reaches a writable connection.
|
|
17
|
+
*
|
|
18
|
+
* ## One day, one transaction
|
|
19
|
+
*
|
|
20
|
+
* `insertDay` is atomic by construction: `BEGIN TRANSACTION`, the day's rows,
|
|
21
|
+
* `COMMIT`. That is what makes the backfill's idempotency hold under a crash -
|
|
22
|
+
* a day is in the table or it is not, so the occupancy check the command runs
|
|
23
|
+
* before each day cannot see half of one and skip the rest. A failure rolls
|
|
24
|
+
* back and propagates: the command stops rather than carrying on to later
|
|
25
|
+
* days, because a run that reported five days inserted while one failed in the
|
|
26
|
+
* middle would leave an operator with no way to tell which history they have.
|
|
27
|
+
*
|
|
28
|
+
* The rows are batched into statements rather than sent one at a time, because
|
|
29
|
+
* a day of a blog's traffic is thousands of rows and one round trip each would
|
|
30
|
+
* make a backfill slower than the log read that feeds it. {@link
|
|
31
|
+
* INSERT_BATCH_ROWS} bounds the statement; every batch is inside the one
|
|
32
|
+
* transaction, so the batching is invisible to a reader of the table.
|
|
33
|
+
*
|
|
34
|
+
* ## Why the column list comes from `schema.ts`
|
|
35
|
+
*
|
|
36
|
+
* Every statement names all twenty columns in `PAGE_VIEWS_COLUMNS`' order and
|
|
37
|
+
* binds a value or a NULL for each, rather than naming only the columns a row
|
|
38
|
+
* happens to carry. Two reasons: a batch's rows do not all carry the same
|
|
39
|
+
* optional columns, so a per-row column list would mean a statement per row;
|
|
40
|
+
* and the cast each value is written through comes from the column's own
|
|
41
|
+
* `icebergType`, so a column added to the table is a column this module writes
|
|
42
|
+
* without being edited - the same property `map-record.ts` has on the read
|
|
43
|
+
* side.
|
|
44
|
+
*/
|
|
45
|
+
import type { CredentialProvider } from 'blogwright-core';
|
|
46
|
+
import type { AnalyticsIngest } from '../ports.js';
|
|
47
|
+
import { type DuckDbConnect, type DuckDbSessionContext } from './duckdb-session.js';
|
|
48
|
+
/** What {@link createDuckDbAnalyticsIngest} is built from. */
|
|
49
|
+
export interface DuckDbAnalyticsIngestOptions {
|
|
50
|
+
/**
|
|
51
|
+
* The plugin context. The session resolves the analytics config from it
|
|
52
|
+
* itself, so no caller can hand this adapter a bucket name that dropped the
|
|
53
|
+
* environment - the same seal the read adapter is held to, and it matters
|
|
54
|
+
* more here: a write against the wrong environment's table cannot be undone
|
|
55
|
+
* by re-running the command.
|
|
56
|
+
*/
|
|
57
|
+
readonly ctx: DuckDbSessionContext;
|
|
58
|
+
/** Credentials for the catalog, resolved through core's provider chain. */
|
|
59
|
+
readonly credentials: CredentialProvider;
|
|
60
|
+
/**
|
|
61
|
+
* How a DuckDB connection is obtained. Defaults to the session's own
|
|
62
|
+
* `connectDuckDb`; a test substitutes a recording connection here.
|
|
63
|
+
*/
|
|
64
|
+
readonly connect?: DuckDbConnect | undefined;
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* Build the DuckDB-backed {@link AnalyticsIngest}. Returns the port and
|
|
68
|
+
* nothing wider, so a caller can insert days and can neither run a statement
|
|
69
|
+
* of its own nor read the table back through it.
|
|
70
|
+
*
|
|
71
|
+
* The connection is opened lazily on the first insert, so constructing this at
|
|
72
|
+
* the plugin's composition root costs nothing on a run that refuses before it
|
|
73
|
+
* reaches the table - which is what the backfill's missing-`createdDay`
|
|
74
|
+
* refusal does.
|
|
75
|
+
*/
|
|
76
|
+
export declare function createDuckDbAnalyticsIngest(opts: DuckDbAnalyticsIngestOptions): AnalyticsIngest;
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The real {@link AnalyticsIngest}: DuckDB with the S3 Tables catalog attached
|
|
3
|
+
* writable, inserting one whole UTC day per transaction. It is the write half
|
|
4
|
+
* of the change spec's §Backfill of historical logs - "written through the
|
|
5
|
+
* DuckDB dependency the dashboard already ships, behind a write port of its
|
|
6
|
+
* own" - and it shares every part of the session with the read adapter beside
|
|
7
|
+
* it through `duckdb-session.ts`: the same credential secret, the same attach
|
|
8
|
+
* target, the same quoting and the same rule that no vendor error object
|
|
9
|
+
* escapes. The one clause that differs is `READ_ONLY`, which this session
|
|
10
|
+
* omits.
|
|
11
|
+
*
|
|
12
|
+
* **The steady-state pipeline does not come through here.** Firehose writes
|
|
13
|
+
* the table itself; this module exists for the one-shot `analytics backfill`
|
|
14
|
+
* action and is constructed only by that command. The dashboard's session is a
|
|
15
|
+
* different object with `readOnly: true`, so nothing the server can be asked
|
|
16
|
+
* to do reaches a writable connection.
|
|
17
|
+
*
|
|
18
|
+
* ## One day, one transaction
|
|
19
|
+
*
|
|
20
|
+
* `insertDay` is atomic by construction: `BEGIN TRANSACTION`, the day's rows,
|
|
21
|
+
* `COMMIT`. That is what makes the backfill's idempotency hold under a crash -
|
|
22
|
+
* a day is in the table or it is not, so the occupancy check the command runs
|
|
23
|
+
* before each day cannot see half of one and skip the rest. A failure rolls
|
|
24
|
+
* back and propagates: the command stops rather than carrying on to later
|
|
25
|
+
* days, because a run that reported five days inserted while one failed in the
|
|
26
|
+
* middle would leave an operator with no way to tell which history they have.
|
|
27
|
+
*
|
|
28
|
+
* The rows are batched into statements rather than sent one at a time, because
|
|
29
|
+
* a day of a blog's traffic is thousands of rows and one round trip each would
|
|
30
|
+
* make a backfill slower than the log read that feeds it. {@link
|
|
31
|
+
* INSERT_BATCH_ROWS} bounds the statement; every batch is inside the one
|
|
32
|
+
* transaction, so the batching is invisible to a reader of the table.
|
|
33
|
+
*
|
|
34
|
+
* ## Why the column list comes from `schema.ts`
|
|
35
|
+
*
|
|
36
|
+
* Every statement names all twenty columns in `PAGE_VIEWS_COLUMNS`' order and
|
|
37
|
+
* binds a value or a NULL for each, rather than naming only the columns a row
|
|
38
|
+
* happens to carry. Two reasons: a batch's rows do not all carry the same
|
|
39
|
+
* optional columns, so a per-row column list would mean a statement per row;
|
|
40
|
+
* and the cast each value is written through comes from the column's own
|
|
41
|
+
* `icebergType`, so a column added to the table is a column this module writes
|
|
42
|
+
* without being edited - the same property `map-record.ts` has on the read
|
|
43
|
+
* side.
|
|
44
|
+
*/
|
|
45
|
+
import { PAGE_VIEWS_COLUMNS } from '../schema.js';
|
|
46
|
+
import { createDuckDbSession, quoteIdentifier, } from './duckdb-session.js';
|
|
47
|
+
/**
|
|
48
|
+
* How many rows one `INSERT` statement carries. Chosen against the statement
|
|
49
|
+
* rather than against the data: twenty columns a row, so five hundred rows is
|
|
50
|
+
* ten thousand bound placeholders - large enough that a day of a blog's
|
|
51
|
+
* traffic is a handful of statements, small enough to stay well inside any
|
|
52
|
+
* parser's limits and to keep one failure's error message readable.
|
|
53
|
+
*/
|
|
54
|
+
const INSERT_BATCH_ROWS = 500;
|
|
55
|
+
/** The SQL type each Iceberg column type is cast to on the way in. */
|
|
56
|
+
const SQL_TYPES = {
|
|
57
|
+
string: 'VARCHAR',
|
|
58
|
+
timestamp: 'TIMESTAMP',
|
|
59
|
+
date: 'DATE',
|
|
60
|
+
int: 'INTEGER',
|
|
61
|
+
long: 'BIGINT',
|
|
62
|
+
double: 'DOUBLE',
|
|
63
|
+
boolean: 'BOOLEAN',
|
|
64
|
+
};
|
|
65
|
+
/** The column list every insert names, in the table's own order. */
|
|
66
|
+
const COLUMN_LIST = PAGE_VIEWS_COLUMNS.map((column) => quoteIdentifier(column.name)).join(', ');
|
|
67
|
+
/**
|
|
68
|
+
* The placeholder one row's column binds to. Row index and column name, so a
|
|
69
|
+
* batch's placeholders are unique and a failure's message points at a row
|
|
70
|
+
* rather than at a position.
|
|
71
|
+
*/
|
|
72
|
+
function placeholder(rowIndex, column) {
|
|
73
|
+
return `r${rowIndex}_${column}`;
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* One batch as a statement and its bindings: every column cast to the type
|
|
77
|
+
* `schema.ts` declares for it, and every absent optional column bound as NULL
|
|
78
|
+
* rather than omitted. An absent value is what "the request had nothing to say
|
|
79
|
+
* for this field" means, and it is exactly what the Firehose path writes for
|
|
80
|
+
* the same record.
|
|
81
|
+
*/
|
|
82
|
+
function insertStatement(relation, rows) {
|
|
83
|
+
const bindings = {};
|
|
84
|
+
const tuples = rows.map((row, rowIndex) => {
|
|
85
|
+
const values = PAGE_VIEWS_COLUMNS.map((column) => {
|
|
86
|
+
const name = placeholder(rowIndex, column.name);
|
|
87
|
+
const value = row[column.name];
|
|
88
|
+
bindings[name] = value === undefined ? null : value;
|
|
89
|
+
return `CAST($${name} AS ${SQL_TYPES[column.icebergType]})`;
|
|
90
|
+
});
|
|
91
|
+
return `(${values.join(', ')})`;
|
|
92
|
+
});
|
|
93
|
+
return {
|
|
94
|
+
sql: `INSERT INTO ${relation} (${COLUMN_LIST}) VALUES ${tuples.join(', ')}`,
|
|
95
|
+
bindings,
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
/** `rows` in chunks of at most {@link INSERT_BATCH_ROWS}. */
|
|
99
|
+
function batched(rows) {
|
|
100
|
+
const batches = [];
|
|
101
|
+
for (let start = 0; start < rows.length; start += INSERT_BATCH_ROWS) {
|
|
102
|
+
batches.push(rows.slice(start, start + INSERT_BATCH_ROWS));
|
|
103
|
+
}
|
|
104
|
+
return batches;
|
|
105
|
+
}
|
|
106
|
+
/**
|
|
107
|
+
* Build the DuckDB-backed {@link AnalyticsIngest}. Returns the port and
|
|
108
|
+
* nothing wider, so a caller can insert days and can neither run a statement
|
|
109
|
+
* of its own nor read the table back through it.
|
|
110
|
+
*
|
|
111
|
+
* The connection is opened lazily on the first insert, so constructing this at
|
|
112
|
+
* the plugin's composition root costs nothing on a run that refuses before it
|
|
113
|
+
* reaches the table - which is what the backfill's missing-`createdDay`
|
|
114
|
+
* refusal does.
|
|
115
|
+
*/
|
|
116
|
+
export function createDuckDbAnalyticsIngest(opts) {
|
|
117
|
+
const session = createDuckDbSession({
|
|
118
|
+
ctx: opts.ctx,
|
|
119
|
+
credentials: opts.credentials,
|
|
120
|
+
readOnly: false,
|
|
121
|
+
connect: opts.connect,
|
|
122
|
+
});
|
|
123
|
+
function contextualise(day, err) {
|
|
124
|
+
return new Error(`analytics ingest of day ${day} into ${session.attachTarget} failed while ${session.detail(err, 'inserting the rows')}`);
|
|
125
|
+
}
|
|
126
|
+
return {
|
|
127
|
+
async insertDay(day, rows) {
|
|
128
|
+
// Both refusals are the port's documented contract, checked before a
|
|
129
|
+
// connection is opened so a caller's mistake costs no AWS round trip.
|
|
130
|
+
if (rows.length === 0) {
|
|
131
|
+
throw new Error(`analytics ingest was asked to insert day ${day} with no rows`);
|
|
132
|
+
}
|
|
133
|
+
const foreign = rows.find((row) => row.day !== day);
|
|
134
|
+
if (foreign !== undefined) {
|
|
135
|
+
throw new Error(`analytics ingest was asked to insert day ${day} carrying a row for day ${foreign.day}`);
|
|
136
|
+
}
|
|
137
|
+
let connection;
|
|
138
|
+
try {
|
|
139
|
+
connection = await session.open();
|
|
140
|
+
}
|
|
141
|
+
catch (err) {
|
|
142
|
+
throw contextualise(day, err);
|
|
143
|
+
}
|
|
144
|
+
try {
|
|
145
|
+
await session.step('beginning the transaction', () => connection.run('BEGIN TRANSACTION', {}));
|
|
146
|
+
}
|
|
147
|
+
catch (err) {
|
|
148
|
+
throw contextualise(day, err);
|
|
149
|
+
}
|
|
150
|
+
try {
|
|
151
|
+
for (const batch of batched(rows)) {
|
|
152
|
+
const statement = insertStatement(session.relation, batch);
|
|
153
|
+
await session.step('inserting the rows', () => connection.run(statement.sql, statement.bindings));
|
|
154
|
+
}
|
|
155
|
+
await session.step('committing the transaction', () => connection.run('COMMIT', {}));
|
|
156
|
+
}
|
|
157
|
+
catch (err) {
|
|
158
|
+
// Best effort, and deliberately not reported: the failure a caller
|
|
159
|
+
// needs to read is the one that ended the insert, not a second one
|
|
160
|
+
// raised while unwinding it. A rollback that itself fails leaves the
|
|
161
|
+
// connection unusable, which is why the command above stops at the
|
|
162
|
+
// first failing day rather than trying the next one on it.
|
|
163
|
+
try {
|
|
164
|
+
await connection.run('ROLLBACK', {});
|
|
165
|
+
}
|
|
166
|
+
catch {
|
|
167
|
+
/* the original failure is the one worth raising */
|
|
168
|
+
}
|
|
169
|
+
throw contextualise(day, err);
|
|
170
|
+
}
|
|
171
|
+
},
|
|
172
|
+
};
|
|
173
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The real {@link AnalyticsQuery}: DuckDB with the S3 Tables catalog attached
|
|
3
|
+
* read-only. Everything about the session - the credential secret, the attach,
|
|
4
|
+
* the quoting and the translation of vendor errors - belongs to
|
|
5
|
+
* `duckdb-session.ts`, which the write adapter beside this one shares; what
|
|
6
|
+
* lives here is only what reading a named query adds to it.
|
|
7
|
+
*
|
|
8
|
+
* **Read-only is a property of the connection, not of this module's shape.**
|
|
9
|
+
* The session is built with `readOnly: true`, so the `ATTACH` carries
|
|
10
|
+
* `READ_ONLY` and DuckDB itself refuses a write on it. `AnalyticsQuery.run` is
|
|
11
|
+
* the whole surface this adapter returns, so there is no statement to supply
|
|
12
|
+
* and no write path to reach even before that.
|
|
13
|
+
*/
|
|
14
|
+
import type { CredentialProvider } from 'blogwright-core';
|
|
15
|
+
import type { AnalyticsQuery } from '../ports.js';
|
|
16
|
+
import { type PreparedQuery } from '../queries.js';
|
|
17
|
+
import { type DuckDbConnect, type DuckDbSessionContext } from './duckdb-session.js';
|
|
18
|
+
/** What {@link createDuckDbAnalyticsQuery} is built from. */
|
|
19
|
+
export interface DuckDbAnalyticsQueryOptions {
|
|
20
|
+
/**
|
|
21
|
+
* The plugin context. The session resolves the analytics config from it
|
|
22
|
+
* itself: `tableBucket` is sealed under task 44's `ENV_DERIVED` symbol and
|
|
23
|
+
* `resolveAnalyticsConfig` is the only way to it, so no caller can hand this
|
|
24
|
+
* adapter a bucket name that dropped the environment.
|
|
25
|
+
*/
|
|
26
|
+
readonly ctx: DuckDbSessionContext;
|
|
27
|
+
/**
|
|
28
|
+
* Credentials for the catalog, resolved through core's provider chain
|
|
29
|
+
* (`createCredentialProvider`). A test passes `staticCredentials`.
|
|
30
|
+
*/
|
|
31
|
+
readonly credentials: CredentialProvider;
|
|
32
|
+
/**
|
|
33
|
+
* How a DuckDB connection is obtained. Defaults to the session's own
|
|
34
|
+
* `connectDuckDb`; a test substitutes a recording connection here.
|
|
35
|
+
*/
|
|
36
|
+
readonly connect?: DuckDbConnect | undefined;
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* The statement a prepared query actually runs as: its definition's SQL with
|
|
40
|
+
* the fixed relation name rewritten to `relation`. Raises when a definition
|
|
41
|
+
* names no relation at all, because a statement that reads nothing would
|
|
42
|
+
* otherwise return an empty result that looks like "no traffic that week".
|
|
43
|
+
*/
|
|
44
|
+
export declare function bindPageViewsRelation(prepared: PreparedQuery, relation: string): string;
|
|
45
|
+
/**
|
|
46
|
+
* Build the DuckDB-backed {@link AnalyticsQuery}. Returns the port and nothing
|
|
47
|
+
* wider: `run(name, params)` is the whole surface, so there is no statement to
|
|
48
|
+
* supply and no write path to reach. Construct it at the plugin's composition
|
|
49
|
+
* root - the dashboard, status and backfill commands - and hand every domain
|
|
50
|
+
* module the port.
|
|
51
|
+
*
|
|
52
|
+
* The connection is opened lazily on the first query and then reused, so
|
|
53
|
+
* building the adapter touches neither the network nor the native library, and
|
|
54
|
+
* the dashboard binds its port before AWS is ever consulted.
|
|
55
|
+
*/
|
|
56
|
+
export declare function createDuckDbAnalyticsQuery(opts: DuckDbAnalyticsQueryOptions): AnalyticsQuery;
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The real {@link AnalyticsQuery}: DuckDB with the S3 Tables catalog attached
|
|
3
|
+
* read-only. Everything about the session - the credential secret, the attach,
|
|
4
|
+
* the quoting and the translation of vendor errors - belongs to
|
|
5
|
+
* `duckdb-session.ts`, which the write adapter beside this one shares; what
|
|
6
|
+
* lives here is only what reading a named query adds to it.
|
|
7
|
+
*
|
|
8
|
+
* **Read-only is a property of the connection, not of this module's shape.**
|
|
9
|
+
* The session is built with `readOnly: true`, so the `ATTACH` carries
|
|
10
|
+
* `READ_ONLY` and DuckDB itself refuses a write on it. `AnalyticsQuery.run` is
|
|
11
|
+
* the whole surface this adapter returns, so there is no statement to supply
|
|
12
|
+
* and no write path to reach even before that.
|
|
13
|
+
*/
|
|
14
|
+
import { PAGE_VIEWS_RELATION, prepareQuery, } from '../queries.js';
|
|
15
|
+
import { createDuckDbSession, } from './duckdb-session.js';
|
|
16
|
+
/**
|
|
17
|
+
* {@link PAGE_VIEWS_RELATION} wherever a statement names it as a whole word.
|
|
18
|
+
* The word boundaries matter: `daily_page_views` must not match, and a bound
|
|
19
|
+
* relation contains the literal `page_views` inside its own quotes, which a
|
|
20
|
+
* replacement callback (rather than a replacement string) leaves alone.
|
|
21
|
+
*/
|
|
22
|
+
const PAGE_VIEWS_RELATION_PATTERN = new RegExp(String.raw `\b${PAGE_VIEWS_RELATION}\b`, 'g');
|
|
23
|
+
/**
|
|
24
|
+
* The statement a prepared query actually runs as: its definition's SQL with
|
|
25
|
+
* the fixed relation name rewritten to `relation`. Raises when a definition
|
|
26
|
+
* names no relation at all, because a statement that reads nothing would
|
|
27
|
+
* otherwise return an empty result that looks like "no traffic that week".
|
|
28
|
+
*/
|
|
29
|
+
export function bindPageViewsRelation(prepared, relation) {
|
|
30
|
+
let bindings = 0;
|
|
31
|
+
const sql = prepared.sql.replaceAll(PAGE_VIEWS_RELATION_PATTERN, () => {
|
|
32
|
+
bindings += 1;
|
|
33
|
+
return relation;
|
|
34
|
+
});
|
|
35
|
+
if (bindings === 0) {
|
|
36
|
+
throw new Error(`analytics query "${prepared.name}" names no ${PAGE_VIEWS_RELATION} relation to bind, so it would read nothing`);
|
|
37
|
+
}
|
|
38
|
+
return sql;
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Build the DuckDB-backed {@link AnalyticsQuery}. Returns the port and nothing
|
|
42
|
+
* wider: `run(name, params)` is the whole surface, so there is no statement to
|
|
43
|
+
* supply and no write path to reach. Construct it at the plugin's composition
|
|
44
|
+
* root - the dashboard, status and backfill commands - and hand every domain
|
|
45
|
+
* module the port.
|
|
46
|
+
*
|
|
47
|
+
* The connection is opened lazily on the first query and then reused, so
|
|
48
|
+
* building the adapter touches neither the network nor the native library, and
|
|
49
|
+
* the dashboard binds its port before AWS is ever consulted.
|
|
50
|
+
*/
|
|
51
|
+
export function createDuckDbAnalyticsQuery(opts) {
|
|
52
|
+
const session = createDuckDbSession({
|
|
53
|
+
ctx: opts.ctx,
|
|
54
|
+
credentials: opts.credentials,
|
|
55
|
+
readOnly: true,
|
|
56
|
+
connect: opts.connect,
|
|
57
|
+
});
|
|
58
|
+
function contextualise(name, err) {
|
|
59
|
+
return new Error(`analytics query "${name}" against ${session.attachTarget} failed while ${session.detail(err, 'running the query')}`);
|
|
60
|
+
}
|
|
61
|
+
return {
|
|
62
|
+
async run(name, params) {
|
|
63
|
+
const prepared = prepareQuery(name, params, session.config);
|
|
64
|
+
const sql = bindPageViewsRelation(prepared, session.relation);
|
|
65
|
+
let connection;
|
|
66
|
+
try {
|
|
67
|
+
connection = await session.open();
|
|
68
|
+
}
|
|
69
|
+
catch (err) {
|
|
70
|
+
throw contextualise(prepared.name, err);
|
|
71
|
+
}
|
|
72
|
+
try {
|
|
73
|
+
return await session.step('executing the statement', () => connection.run(sql, prepared.bindings));
|
|
74
|
+
}
|
|
75
|
+
catch (err) {
|
|
76
|
+
throw contextualise(prepared.name, err);
|
|
77
|
+
}
|
|
78
|
+
},
|
|
79
|
+
};
|
|
80
|
+
}
|