blogwright-analytics 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/README.md +162 -0
  2. package/dist/adapters/duckdb-ingest.d.ts +76 -0
  3. package/dist/adapters/duckdb-ingest.js +173 -0
  4. package/dist/adapters/duckdb-query.d.ts +56 -0
  5. package/dist/adapters/duckdb-query.js +80 -0
  6. package/dist/adapters/duckdb-session.d.ts +168 -0
  7. package/dist/adapters/duckdb-session.js +330 -0
  8. package/dist/app/_app/immutable/assets/0.BTQrrh5B.css +1 -0
  9. package/dist/app/_app/immutable/assets/2.CZSK3rT8.css +1 -0
  10. package/dist/app/_app/immutable/assets/BrushContext.D7c8UPey.css +1 -0
  11. package/dist/app/_app/immutable/assets/ChartAnnotations.CPxIG7Mw.css +1 -0
  12. package/dist/app/_app/immutable/assets/Circle.C5MKzgk2.css +1 -0
  13. package/dist/app/_app/immutable/assets/DefaultTooltip.C5-uctZ7.css +1 -0
  14. package/dist/app/_app/immutable/assets/Group.DV48xipa.css +1 -0
  15. package/dist/app/_app/immutable/assets/Labels.BxZ4NUVz.css +1 -0
  16. package/dist/app/_app/immutable/assets/Legend.CxnrE4Ye.css +1 -0
  17. package/dist/app/_app/immutable/assets/Line.fkmsECm9.css +1 -0
  18. package/dist/app/_app/immutable/assets/Path.CvpwNZ6g.css +1 -0
  19. package/dist/app/_app/immutable/assets/Rect.CtRaGMmQ.css +1 -0
  20. package/dist/app/_app/immutable/assets/Text.j9l35qB0.css +1 -0
  21. package/dist/app/_app/immutable/assets/TransformContext.Bs_HkpAk.css +1 -0
  22. package/dist/app/_app/immutable/assets/Voronoi.ce7atosu.css +1 -0
  23. package/dist/app/_app/immutable/chunks/-aNGNaBT.js +1 -0
  24. package/dist/app/_app/immutable/chunks/6djn-yLs.js +1 -0
  25. package/dist/app/_app/immutable/chunks/B1amyutE.js +1 -0
  26. package/dist/app/_app/immutable/chunks/B3vZDoek.js +1 -0
  27. package/dist/app/_app/immutable/chunks/B5KRA4hC.js +1 -0
  28. package/dist/app/_app/immutable/chunks/BClnVG6H.js +1 -0
  29. package/dist/app/_app/immutable/chunks/BID1NNRh.js +1 -0
  30. package/dist/app/_app/immutable/chunks/BR2LaRms.js +1 -0
  31. package/dist/app/_app/immutable/chunks/Bd1gDe3Y.js +1 -0
  32. package/dist/app/_app/immutable/chunks/Bjy-W4x2.js +81 -0
  33. package/dist/app/_app/immutable/chunks/Bl052uUt.js +1 -0
  34. package/dist/app/_app/immutable/chunks/Bye3lL0c.js +1 -0
  35. package/dist/app/_app/immutable/chunks/C58PZtCD.js +4 -0
  36. package/dist/app/_app/immutable/chunks/CAzydqEO.js +1 -0
  37. package/dist/app/_app/immutable/chunks/CCch3uox.js +1 -0
  38. package/dist/app/_app/immutable/chunks/CIlSMUH9.js +1 -0
  39. package/dist/app/_app/immutable/chunks/CO1vUXfR.js +1 -0
  40. package/dist/app/_app/immutable/chunks/CPbD8C65.js +5 -0
  41. package/dist/app/_app/immutable/chunks/CRTcXoMo.js +1 -0
  42. package/dist/app/_app/immutable/chunks/CjjyIQAO.js +1 -0
  43. package/dist/app/_app/immutable/chunks/CuXAxjvF.js +1 -0
  44. package/dist/app/_app/immutable/chunks/CvyVA_jC.js +1 -0
  45. package/dist/app/_app/immutable/chunks/CxGCFVdy.js +1 -0
  46. package/dist/app/_app/immutable/chunks/D0Ty6LN0.js +1 -0
  47. package/dist/app/_app/immutable/chunks/D2AaQUUW.js +1 -0
  48. package/dist/app/_app/immutable/chunks/D2BnX0Uk.js +3 -0
  49. package/dist/app/_app/immutable/chunks/DJc8C0NK.js +1 -0
  50. package/dist/app/_app/immutable/chunks/DKMlMI4a.js +1 -0
  51. package/dist/app/_app/immutable/chunks/DVXZkpbf.js +1 -0
  52. package/dist/app/_app/immutable/chunks/DVt8ukQ_.js +1 -0
  53. package/dist/app/_app/immutable/chunks/DZPlYdq_.js +1 -0
  54. package/dist/app/_app/immutable/chunks/Db0q5_zr.js +1 -0
  55. package/dist/app/_app/immutable/chunks/Dfvzj6n2.js +1 -0
  56. package/dist/app/_app/immutable/chunks/Dh958be7.js +1 -0
  57. package/dist/app/_app/immutable/chunks/DjKLLdnY.js +15 -0
  58. package/dist/app/_app/immutable/chunks/Doz7YX1W.js +1 -0
  59. package/dist/app/_app/immutable/chunks/DthYhn6Y.js +2 -0
  60. package/dist/app/_app/immutable/chunks/DtuTIrAM.js +1 -0
  61. package/dist/app/_app/immutable/chunks/HclGiUj8.js +1 -0
  62. package/dist/app/_app/immutable/chunks/Hx0TNsV3.js +1 -0
  63. package/dist/app/_app/immutable/chunks/RobXhXPM.js +1 -0
  64. package/dist/app/_app/immutable/chunks/V9ZjaxiY.js +1 -0
  65. package/dist/app/_app/immutable/chunks/Y5urAfNy.js +1 -0
  66. package/dist/app/_app/immutable/chunks/caXkbKD3.js +1 -0
  67. package/dist/app/_app/immutable/chunks/devYm2ud.js +1 -0
  68. package/dist/app/_app/immutable/chunks/mtZWP0zR.js +1 -0
  69. package/dist/app/_app/immutable/chunks/vDgBJUjM.js +1 -0
  70. package/dist/app/_app/immutable/chunks/xIq_fFFM.js +1 -0
  71. package/dist/app/_app/immutable/chunks/xihTtKlq.js +1 -0
  72. package/dist/app/_app/immutable/chunks/z05MoCFz.js +1 -0
  73. package/dist/app/_app/immutable/entry/app.CLAerUAN.js +2 -0
  74. package/dist/app/_app/immutable/entry/start.D3MqnNci.js +1 -0
  75. package/dist/app/_app/immutable/nodes/0.UTMEigHJ.js +1 -0
  76. package/dist/app/_app/immutable/nodes/1.Cn4f11bT.js +1 -0
  77. package/dist/app/_app/immutable/nodes/2.B39cIcr2.js +6 -0
  78. package/dist/app/_app/version.json +1 -0
  79. package/dist/app/index.html +82 -0
  80. package/dist/aws/clients.d.ts +70 -0
  81. package/dist/aws/clients.js +52 -0
  82. package/dist/aws/errors.d.ts +41 -0
  83. package/dist/aws/errors.js +70 -0
  84. package/dist/aws/firehose.d.ts +228 -0
  85. package/dist/aws/firehose.js +347 -0
  86. package/dist/aws/glue.d.ts +103 -0
  87. package/dist/aws/glue.js +225 -0
  88. package/dist/aws/lambda.d.ts +132 -0
  89. package/dist/aws/lambda.js +339 -0
  90. package/dist/aws/s3tables.d.ts +120 -0
  91. package/dist/aws/s3tables.js +281 -0
  92. package/dist/backfill.d.ts +100 -0
  93. package/dist/backfill.js +294 -0
  94. package/dist/commands.d.ts +124 -0
  95. package/dist/commands.js +336 -0
  96. package/dist/config.d.ts +162 -0
  97. package/dist/config.js +317 -0
  98. package/dist/fixture-ingest.d.ts +49 -0
  99. package/dist/fixture-ingest.js +43 -0
  100. package/dist/fixture-query.d.ts +39 -0
  101. package/dist/fixture-query.js +70 -0
  102. package/dist/index.d.ts +35 -0
  103. package/dist/index.js +35 -0
  104. package/dist/nodes.d.ts +404 -0
  105. package/dist/nodes.js +2708 -0
  106. package/dist/paths.d.ts +45 -0
  107. package/dist/paths.js +47 -0
  108. package/dist/plugin.d.ts +102 -0
  109. package/dist/plugin.js +248 -0
  110. package/dist/ports.d.ts +113 -0
  111. package/dist/ports.js +35 -0
  112. package/dist/queries.d.ts +301 -0
  113. package/dist/queries.js +414 -0
  114. package/dist/schema.d.ts +240 -0
  115. package/dist/schema.js +154 -0
  116. package/dist/server.d.ts +150 -0
  117. package/dist/server.js +499 -0
  118. package/dist/transform/bots.d.ts +47 -0
  119. package/dist/transform/bots.js +73 -0
  120. package/dist/transform/handler.d.ts +135 -0
  121. package/dist/transform/handler.js +177 -0
  122. package/dist/transform/map-record.d.ts +110 -0
  123. package/dist/transform/map-record.js +275 -0
  124. package/dist/transform/visitor-key.d.ts +83 -0
  125. package/dist/transform/visitor-key.js +120 -0
  126. package/dist/transform-bundle/index.mjs +21456 -0
  127. package/dist/transform-bundle/transform-manifest.json +4 -0
  128. package/dist/transform-hash.d.ts +135 -0
  129. package/dist/transform-hash.js +186 -0
  130. package/dist/write-transform-manifest.mjs +365 -0
  131. package/package.json +59 -0
package/README.md ADDED
@@ -0,0 +1,162 @@
1
+ # blogwright-analytics
2
+
3
+ Traffic analytics for [blogwright](https://github.com/antstanley/blogwright): a second
4
+ CloudFront access-log delivery routed through Amazon Data Firehose into an Apache Iceberg
5
+ table in an S3 Tables bucket, with a local SvelteKit dashboard that reads that table
6
+ through DuckDB. The site's existing CloudWatch log delivery is untouched - one CloudFront
7
+ delivery source carries several deliveries, so this one is added beside it.
8
+
9
+ This package owns the whole pipeline: its own four AWS service clients (S3 Tables,
10
+ Firehose, Glue, Lambda), the twelve resource nodes those clients reconcile, the
11
+ record-transform Lambda, the named query set, and the dashboard application. It depends
12
+ on `blogwright-core` (ports, the SigV4 transport and signer, the S3 and Secrets Manager
13
+ clients) and on DuckDB, which is reached only through the `AnalyticsQuery` port's one
14
+ adapter. It never imports the CLI.
15
+
16
+ ## Install
17
+
18
+ The plugin is not shipped with the CLI. A repo that never installs it pays nothing:
19
+
20
+ ```sh
21
+ blogwright plugin add analytics
22
+ ```
23
+
24
+ `analytics` resolves to the package `blogwright-analytics`, installed at the running
25
+ CLI's own version and pinned exactly, so the CLI and its plugins cannot drift apart
26
+ between two checkouts of the same repo. `blogwright plugin list` shows it once installed,
27
+ and `blogwright plugin remove analytics` offers to tear its resources down first.
28
+
29
+ ## Commands
30
+
31
+ Once installed, the plugin answers `blogwright analytics <action> [env]`. The environment
32
+ defaults to `production` and `--env` overrides the positional, exactly as for a built-in
33
+ command. Five actions are the steady state:
34
+
35
+ | Action | What it does |
36
+ | --- | --- |
37
+ | `blogwright analytics init [env]` | Asks for the `analytics` config block - namespace, table, bot handling, dashboard port - and splices it into `config/<env>.jsonc`. `blogwright init` asks the same questions as part of the first-run wizard. |
38
+ | `blogwright analytics bootstrap [env]` | Provisions the pipeline: the S3 Tables bucket, its namespace and the `page_views` table, the Glue `s3tablescatalog` federation, the `visitor_key` salt secret, the transform Lambda and its execution role, the Firehose error bucket, delivery role and delivery stream, and the CloudWatch delivery destination and delivery. |
39
+ | `blogwright analytics status [env]` | Reports each node present or missing, then the Firehose stream's delivery health and the table's current row count. An environment that was never bootstrapped is not an error: every node reports missing and the command exits 0. |
40
+ | `blogwright analytics dashboard [env]` | Serves the prebuilt dashboard from `dist/app` on `127.0.0.1` (port 4317 by default) and answers its named queries against the table. |
41
+ | `blogwright analytics destroy [env] --yes` | Removes the plugin's own resources. The Glue federation is the one exception and is never deleted - see below. |
42
+
43
+ Two things the table does not say. The plugin's resources are recorded in their own state
44
+ object, `state/<env>.analytics.json`: `blogwright bootstrap` provisions none of them, and
45
+ `blogwright destroy --yes` refuses while that object exists and names
46
+ `blogwright analytics destroy <env> --yes` as the way through. And the dashboard's server
47
+ answers only the queries this package defines by name - the seven the dashboard charts
48
+ (`views-over-time`, `unique-visitors`, `top-paths`, `referrers`, `countries`,
49
+ `status-codes`, `cache-hit-ratio`) plus the `row-count` that `analytics status` reads.
50
+ It never executes SQL supplied over its socket, and it binds loopback only.
51
+
52
+ ### `blogwright analytics backfill [env]` - optional, one-shot
53
+
54
+ A sixth action, deliberately not in the table above, because it is not part of
55
+ the steady state. Firehose only carries what CloudFront produced after its
56
+ delivery existed; `backfill` is the hand-run pull of the history that came
57
+ before it, and once it has run there is no reason to run it again.
58
+
59
+ It reads the CloudWatch log group the site's own delivery already writes -
60
+ `/<siteName>/<env>/cloudfront`, bounded by `retention.cloudfrontDays` - and
61
+ maps every event through the same code the transform Lambda runs, so a record
62
+ produces the same `page_views` row whichever path carried it, `visitor_key`
63
+ included: the day's salt is `HMAC-SHA256(secret, day)` over the same stored
64
+ secret, so a historical day's salt is derivable and the raw IP is no more
65
+ stored here than it is there.
66
+
67
+ **It cannot double-count, and not by de-duplicating.** The
68
+ `analytics-log-delivery` node records the UTC day it first created the
69
+ delivery, once and never again, and the backfill inserts only whole days
70
+ *strictly before* that day - Firehose received nothing before its delivery
71
+ existed, so the two paths never write the same day. Within that range each day
72
+ is one transaction, a day the table already holds rows for is skipped, and a
73
+ row whose own `day` is not the day being written is not inserted. So a re-run
74
+ inserts nothing, and a run that crashed resumes where it stopped.
75
+
76
+ The boundary day itself is never backfilled. Up to one day of history at the
77
+ seam is lost, which is the accepted precision limit rather than an oversight:
78
+ buying it back would mean comparing rows, and comparing rows is the thing this
79
+ design does not do.
80
+
81
+ The command refuses, before it calls AWS at all, when the plugin's state
82
+ carries no delivery record - run `blogwright analytics bootstrap <env>` first -
83
+ and also when it carries a delivery with no recorded day, which is what a state
84
+ file that lost the key looks like. There is no default in that case: assuming
85
+ "everything" would insert days Firehose has already delivered and double every
86
+ row in them, so the command says what to supply instead. Its report names every
87
+ day it inserted, every day it skipped and why, and the boundary day it left
88
+ alone.
89
+
90
+ ## Everything is created in us-east-1
91
+
92
+ Every resource this plugin owns is created in `us-east-1` regardless of `config.region`,
93
+ and every node's title says so rather than diverging silently.
94
+
95
+ CloudFront forces it. CloudFront is a global service whose logging control plane lives in
96
+ `us-east-1` alone, and standard logging accepts a Firehose delivery stream only there, so
97
+ the stream has to be in that region. Everything the stream reaches has to follow: the
98
+ table bucket and its Iceberg table, the Glue federation the stream writes through, the
99
+ transform Lambda the stream invokes, the Firehose error bucket, and the salt secret the
100
+ Lambda reads. The two IAM roles are global, and state the pipeline they serve instead.
101
+
102
+ That is why this package builds its own S3 and Secrets Manager clients over the host's
103
+ `signingUsEast1` signer rather than reusing `ctx.clients.s3` and `ctx.clients.secrets`:
104
+ the host's pair signs in `config.region`, which would put the error bucket, and the salt,
105
+ in a region the transform function cannot read from.
106
+
107
+ ## Privacy
108
+
109
+ **The raw viewer IP is never stored.** `c-ip` is selected from CloudFront only so the
110
+ transform Lambda can derive `visitor_key` from it, and no column of the `page_views` table
111
+ holds it: it has no entry in the field-to-column map, and the transform discards it after
112
+ hashing. `visitor_key` is a SHA-256 digest over the viewer IP, the user agent and that
113
+ day's salt, and the salt is `HMAC-SHA256(secret, day)` over one long-lived random secret
114
+ held in Secrets Manager - never the date alone, which anyone holding the table could
115
+ compute and then brute-force back across a 32-bit address space. The consequence is
116
+ deliberate and stated where it matters: a `visitor_key` is not comparable across days, so
117
+ a monthly unique-visitor figure is the sum of daily uniques rather than a distinct count.
118
+
119
+ **`cs(Cookie)` and `x-forwarded-for` are never selected**, so they never leave CloudFront
120
+ for this pipeline: they reach neither Firehose, nor the transform, nor the table. This
121
+ governs the analytics delivery only. The site's existing CloudWatch delivery is created
122
+ with no field list, so AWS's default set - which includes both - still applies to that
123
+ copy; narrowing it is a change to the site's own node, not this one.
124
+
125
+ No cookie is set and no identifier is written to a visitor's browser. Bot traffic is
126
+ flagged rather than dropped (`is_bot` is a column, and filtering is a query default), so
127
+ a heuristic that turns out to be wrong is a query change and not lost data.
128
+
129
+ ## Shared state, and what teardown leaves behind
130
+
131
+ One resource is account-and-region scoped rather than per-environment: the Glue
132
+ `s3tablescatalog` federation that Firehose reaches S3 Tables through. Two environments of
133
+ the same site share it. Its node therefore adopts an existing federation rather than
134
+ failing, and its `delete()` is a no-op - tearing down staging must not break production,
135
+ and the Glue API this package speaks exposes no delete operation at all. Everything else
136
+ the plugin creates is per-environment and is removed by
137
+ `blogwright analytics destroy <env> --yes`.
138
+
139
+ Rows are never aged out. The table is append-only and partitioned by `day`, so expiring
140
+ old data would mean whole-partition deletes issued on a schedule; S3 Tables offers no
141
+ row-retention setting for a table you create. The site's `retention.cloudfrontDays`
142
+ governs only the CloudWatch copy of the logs.
143
+
144
+ ## Configuration
145
+
146
+ The `analytics` block in `config/<env>.jsonc`, all of it optional:
147
+
148
+ | Key | Default | Meaning |
149
+ | --- | --- | --- |
150
+ | `namespace` | `web` | Iceberg namespace holding the table. |
151
+ | `table` | `page_views` | Iceberg table the page views land in. |
152
+ | `bots` | `flag` | `flag` keeps bot rows and marks them; `filter` excludes them from queries. |
153
+ | `dashboard.port` | `4317` | Port the local dashboard binds on `127.0.0.1`. |
154
+ | `tableBucket` | `<env>-<siteName>-analytics` | S3 Tables bucket holding the namespace. |
155
+ | `saltSecretName` | `<siteName>/<env>/analytics-salt` | Secrets Manager secret the daily salt is derived from. |
156
+
157
+ The last two carry the environment in their defaults and are not asked by the wizard: a
158
+ prompt whose default is wrong for every environment but one is worse than no prompt. Write
159
+ them into the block by hand to override either, where the same validator still checks
160
+ them. Do not take the environment out of either default: without it two environments
161
+ resolve to the same Iceberg table and the same salt, and `blogwright analytics destroy`
162
+ in staging would delete production's data.
@@ -0,0 +1,76 @@
1
+ /**
2
+ * The real {@link AnalyticsIngest}: DuckDB with the S3 Tables catalog attached
3
+ * writable, inserting one whole UTC day per transaction. It is the write half
4
+ * of the change spec's §Backfill of historical logs - "written through the
5
+ * DuckDB dependency the dashboard already ships, behind a write port of its
6
+ * own" - and it shares every part of the session with the read adapter beside
7
+ * it through `duckdb-session.ts`: the same credential secret, the same attach
8
+ * target, the same quoting and the same rule that no vendor error object
9
+ * escapes. The one clause that differs is `READ_ONLY`, which this session
10
+ * omits.
11
+ *
12
+ * **The steady-state pipeline does not come through here.** Firehose writes
13
+ * the table itself; this module exists for the one-shot `analytics backfill`
14
+ * action and is constructed only by that command. The dashboard's session is a
15
+ * different object with `readOnly: true`, so nothing the server can be asked
16
+ * to do reaches a writable connection.
17
+ *
18
+ * ## One day, one transaction
19
+ *
20
+ * `insertDay` is atomic by construction: `BEGIN TRANSACTION`, the day's rows,
21
+ * `COMMIT`. That is what makes the backfill's idempotency hold under a crash -
22
+ * a day is in the table or it is not, so the occupancy check the command runs
23
+ * before each day cannot see half of one and skip the rest. A failure rolls
24
+ * back and propagates: the command stops rather than carrying on to later
25
+ * days, because a run that reported five days inserted while one failed in the
26
+ * middle would leave an operator with no way to tell which history they have.
27
+ *
28
+ * The rows are batched into statements rather than sent one at a time, because
29
+ * a day of a blog's traffic is thousands of rows and one round trip each would
30
+ * make a backfill slower than the log read that feeds it. {@link
31
+ * INSERT_BATCH_ROWS} bounds the statement; every batch is inside the one
32
+ * transaction, so the batching is invisible to a reader of the table.
33
+ *
34
+ * ## Why the column list comes from `schema.ts`
35
+ *
36
+ * Every statement names all twenty columns in `PAGE_VIEWS_COLUMNS`' order and
37
+ * binds a value or a NULL for each, rather than naming only the columns a row
38
+ * happens to carry. Two reasons: a batch's rows do not all carry the same
39
+ * optional columns, so a per-row column list would mean a statement per row;
40
+ * and the cast each value is written through comes from the column's own
41
+ * `icebergType`, so a column added to the table is a column this module writes
42
+ * without being edited - the same property `map-record.ts` has on the read
43
+ * side.
44
+ */
45
+ import type { CredentialProvider } from 'blogwright-core';
46
+ import type { AnalyticsIngest } from '../ports.js';
47
+ import { type DuckDbConnect, type DuckDbSessionContext } from './duckdb-session.js';
48
+ /** What {@link createDuckDbAnalyticsIngest} is built from. */
49
+ export interface DuckDbAnalyticsIngestOptions {
50
+ /**
51
+ * The plugin context. The session resolves the analytics config from it
52
+ * itself, so no caller can hand this adapter a bucket name that dropped the
53
+ * environment - the same seal the read adapter is held to, and it matters
54
+ * more here: a write against the wrong environment's table cannot be undone
55
+ * by re-running the command.
56
+ */
57
+ readonly ctx: DuckDbSessionContext;
58
+ /** Credentials for the catalog, resolved through core's provider chain. */
59
+ readonly credentials: CredentialProvider;
60
+ /**
61
+ * How a DuckDB connection is obtained. Defaults to the session's own
62
+ * `connectDuckDb`; a test substitutes a recording connection here.
63
+ */
64
+ readonly connect?: DuckDbConnect | undefined;
65
+ }
66
+ /**
67
+ * Build the DuckDB-backed {@link AnalyticsIngest}. Returns the port and
68
+ * nothing wider, so a caller can insert days and can neither run a statement
69
+ * of its own nor read the table back through it.
70
+ *
71
+ * The connection is opened lazily on the first insert, so constructing this at
72
+ * the plugin's composition root costs nothing on a run that refuses before it
73
+ * reaches the table - which is what the backfill's missing-`createdDay`
74
+ * refusal does.
75
+ */
76
+ export declare function createDuckDbAnalyticsIngest(opts: DuckDbAnalyticsIngestOptions): AnalyticsIngest;
@@ -0,0 +1,173 @@
1
+ /**
2
+ * The real {@link AnalyticsIngest}: DuckDB with the S3 Tables catalog attached
3
+ * writable, inserting one whole UTC day per transaction. It is the write half
4
+ * of the change spec's §Backfill of historical logs - "written through the
5
+ * DuckDB dependency the dashboard already ships, behind a write port of its
6
+ * own" - and it shares every part of the session with the read adapter beside
7
+ * it through `duckdb-session.ts`: the same credential secret, the same attach
8
+ * target, the same quoting and the same rule that no vendor error object
9
+ * escapes. The one clause that differs is `READ_ONLY`, which this session
10
+ * omits.
11
+ *
12
+ * **The steady-state pipeline does not come through here.** Firehose writes
13
+ * the table itself; this module exists for the one-shot `analytics backfill`
14
+ * action and is constructed only by that command. The dashboard's session is a
15
+ * different object with `readOnly: true`, so nothing the server can be asked
16
+ * to do reaches a writable connection.
17
+ *
18
+ * ## One day, one transaction
19
+ *
20
+ * `insertDay` is atomic by construction: `BEGIN TRANSACTION`, the day's rows,
21
+ * `COMMIT`. That is what makes the backfill's idempotency hold under a crash -
22
+ * a day is in the table or it is not, so the occupancy check the command runs
23
+ * before each day cannot see half of one and skip the rest. A failure rolls
24
+ * back and propagates: the command stops rather than carrying on to later
25
+ * days, because a run that reported five days inserted while one failed in the
26
+ * middle would leave an operator with no way to tell which history they have.
27
+ *
28
+ * The rows are batched into statements rather than sent one at a time, because
29
+ * a day of a blog's traffic is thousands of rows and one round trip each would
30
+ * make a backfill slower than the log read that feeds it. {@link
31
+ * INSERT_BATCH_ROWS} bounds the statement; every batch is inside the one
32
+ * transaction, so the batching is invisible to a reader of the table.
33
+ *
34
+ * ## Why the column list comes from `schema.ts`
35
+ *
36
+ * Every statement names all twenty columns in `PAGE_VIEWS_COLUMNS`' order and
37
+ * binds a value or a NULL for each, rather than naming only the columns a row
38
+ * happens to carry. Two reasons: a batch's rows do not all carry the same
39
+ * optional columns, so a per-row column list would mean a statement per row;
40
+ * and the cast each value is written through comes from the column's own
41
+ * `icebergType`, so a column added to the table is a column this module writes
42
+ * without being edited - the same property `map-record.ts` has on the read
43
+ * side.
44
+ */
45
+ import { PAGE_VIEWS_COLUMNS } from '../schema.js';
46
+ import { createDuckDbSession, quoteIdentifier, } from './duckdb-session.js';
47
+ /**
48
+ * How many rows one `INSERT` statement carries. Chosen against the statement
49
+ * rather than against the data: twenty columns a row, so five hundred rows is
50
+ * ten thousand bound placeholders - large enough that a day of a blog's
51
+ * traffic is a handful of statements, small enough to stay well inside any
52
+ * parser's limits and to keep one failure's error message readable.
53
+ */
54
+ const INSERT_BATCH_ROWS = 500;
55
+ /** The SQL type each Iceberg column type is cast to on the way in. */
56
+ const SQL_TYPES = {
57
+ string: 'VARCHAR',
58
+ timestamp: 'TIMESTAMP',
59
+ date: 'DATE',
60
+ int: 'INTEGER',
61
+ long: 'BIGINT',
62
+ double: 'DOUBLE',
63
+ boolean: 'BOOLEAN',
64
+ };
65
+ /** The column list every insert names, in the table's own order. */
66
+ const COLUMN_LIST = PAGE_VIEWS_COLUMNS.map((column) => quoteIdentifier(column.name)).join(', ');
67
+ /**
68
+ * The placeholder one row's column binds to. Row index and column name, so a
69
+ * batch's placeholders are unique and a failure's message points at a row
70
+ * rather than at a position.
71
+ */
72
+ function placeholder(rowIndex, column) {
73
+ return `r${rowIndex}_${column}`;
74
+ }
75
+ /**
76
+ * One batch as a statement and its bindings: every column cast to the type
77
+ * `schema.ts` declares for it, and every absent optional column bound as NULL
78
+ * rather than omitted. An absent value is what "the request had nothing to say
79
+ * for this field" means, and it is exactly what the Firehose path writes for
80
+ * the same record.
81
+ */
82
+ function insertStatement(relation, rows) {
83
+ const bindings = {};
84
+ const tuples = rows.map((row, rowIndex) => {
85
+ const values = PAGE_VIEWS_COLUMNS.map((column) => {
86
+ const name = placeholder(rowIndex, column.name);
87
+ const value = row[column.name];
88
+ bindings[name] = value === undefined ? null : value;
89
+ return `CAST($${name} AS ${SQL_TYPES[column.icebergType]})`;
90
+ });
91
+ return `(${values.join(', ')})`;
92
+ });
93
+ return {
94
+ sql: `INSERT INTO ${relation} (${COLUMN_LIST}) VALUES ${tuples.join(', ')}`,
95
+ bindings,
96
+ };
97
+ }
98
+ /** `rows` in chunks of at most {@link INSERT_BATCH_ROWS}. */
99
+ function batched(rows) {
100
+ const batches = [];
101
+ for (let start = 0; start < rows.length; start += INSERT_BATCH_ROWS) {
102
+ batches.push(rows.slice(start, start + INSERT_BATCH_ROWS));
103
+ }
104
+ return batches;
105
+ }
106
+ /**
107
+ * Build the DuckDB-backed {@link AnalyticsIngest}. Returns the port and
108
+ * nothing wider, so a caller can insert days and can neither run a statement
109
+ * of its own nor read the table back through it.
110
+ *
111
+ * The connection is opened lazily on the first insert, so constructing this at
112
+ * the plugin's composition root costs nothing on a run that refuses before it
113
+ * reaches the table - which is what the backfill's missing-`createdDay`
114
+ * refusal does.
115
+ */
116
+ export function createDuckDbAnalyticsIngest(opts) {
117
+ const session = createDuckDbSession({
118
+ ctx: opts.ctx,
119
+ credentials: opts.credentials,
120
+ readOnly: false,
121
+ connect: opts.connect,
122
+ });
123
+ function contextualise(day, err) {
124
+ return new Error(`analytics ingest of day ${day} into ${session.attachTarget} failed while ${session.detail(err, 'inserting the rows')}`);
125
+ }
126
+ return {
127
+ async insertDay(day, rows) {
128
+ // Both refusals are the port's documented contract, checked before a
129
+ // connection is opened so a caller's mistake costs no AWS round trip.
130
+ if (rows.length === 0) {
131
+ throw new Error(`analytics ingest was asked to insert day ${day} with no rows`);
132
+ }
133
+ const foreign = rows.find((row) => row.day !== day);
134
+ if (foreign !== undefined) {
135
+ throw new Error(`analytics ingest was asked to insert day ${day} carrying a row for day ${foreign.day}`);
136
+ }
137
+ let connection;
138
+ try {
139
+ connection = await session.open();
140
+ }
141
+ catch (err) {
142
+ throw contextualise(day, err);
143
+ }
144
+ try {
145
+ await session.step('beginning the transaction', () => connection.run('BEGIN TRANSACTION', {}));
146
+ }
147
+ catch (err) {
148
+ throw contextualise(day, err);
149
+ }
150
+ try {
151
+ for (const batch of batched(rows)) {
152
+ const statement = insertStatement(session.relation, batch);
153
+ await session.step('inserting the rows', () => connection.run(statement.sql, statement.bindings));
154
+ }
155
+ await session.step('committing the transaction', () => connection.run('COMMIT', {}));
156
+ }
157
+ catch (err) {
158
+ // Best effort, and deliberately not reported: the failure a caller
159
+ // needs to read is the one that ended the insert, not a second one
160
+ // raised while unwinding it. A rollback that itself fails leaves the
161
+ // connection unusable, which is why the command above stops at the
162
+ // first failing day rather than trying the next one on it.
163
+ try {
164
+ await connection.run('ROLLBACK', {});
165
+ }
166
+ catch {
167
+ /* the original failure is the one worth raising */
168
+ }
169
+ throw contextualise(day, err);
170
+ }
171
+ },
172
+ };
173
+ }
@@ -0,0 +1,56 @@
1
+ /**
2
+ * The real {@link AnalyticsQuery}: DuckDB with the S3 Tables catalog attached
3
+ * read-only. Everything about the session - the credential secret, the attach,
4
+ * the quoting and the translation of vendor errors - belongs to
5
+ * `duckdb-session.ts`, which the write adapter beside this one shares; what
6
+ * lives here is only what reading a named query adds to it.
7
+ *
8
+ * **Read-only is a property of the connection, not of this module's shape.**
9
+ * The session is built with `readOnly: true`, so the `ATTACH` carries
10
+ * `READ_ONLY` and DuckDB itself refuses a write on it. `AnalyticsQuery.run` is
11
+ * the whole surface this adapter returns, so there is no statement to supply
12
+ * and no write path to reach even before that.
13
+ */
14
+ import type { CredentialProvider } from 'blogwright-core';
15
+ import type { AnalyticsQuery } from '../ports.js';
16
+ import { type PreparedQuery } from '../queries.js';
17
+ import { type DuckDbConnect, type DuckDbSessionContext } from './duckdb-session.js';
18
+ /** What {@link createDuckDbAnalyticsQuery} is built from. */
19
+ export interface DuckDbAnalyticsQueryOptions {
20
+ /**
21
+ * The plugin context. The session resolves the analytics config from it
22
+ * itself: `tableBucket` is sealed under task 44's `ENV_DERIVED` symbol and
23
+ * `resolveAnalyticsConfig` is the only way to it, so no caller can hand this
24
+ * adapter a bucket name that dropped the environment.
25
+ */
26
+ readonly ctx: DuckDbSessionContext;
27
+ /**
28
+ * Credentials for the catalog, resolved through core's provider chain
29
+ * (`createCredentialProvider`). A test passes `staticCredentials`.
30
+ */
31
+ readonly credentials: CredentialProvider;
32
+ /**
33
+ * How a DuckDB connection is obtained. Defaults to the session's own
34
+ * `connectDuckDb`; a test substitutes a recording connection here.
35
+ */
36
+ readonly connect?: DuckDbConnect | undefined;
37
+ }
38
+ /**
39
+ * The statement a prepared query actually runs as: its definition's SQL with
40
+ * the fixed relation name rewritten to `relation`. Raises when a definition
41
+ * names no relation at all, because a statement that reads nothing would
42
+ * otherwise return an empty result that looks like "no traffic that week".
43
+ */
44
+ export declare function bindPageViewsRelation(prepared: PreparedQuery, relation: string): string;
45
+ /**
46
+ * Build the DuckDB-backed {@link AnalyticsQuery}. Returns the port and nothing
47
+ * wider: `run(name, params)` is the whole surface, so there is no statement to
48
+ * supply and no write path to reach. Construct it at the plugin's composition
49
+ * root - the dashboard, status and backfill commands - and hand every domain
50
+ * module the port.
51
+ *
52
+ * The connection is opened lazily on the first query and then reused, so
53
+ * building the adapter touches neither the network nor the native library, and
54
+ * the dashboard binds its port before AWS is ever consulted.
55
+ */
56
+ export declare function createDuckDbAnalyticsQuery(opts: DuckDbAnalyticsQueryOptions): AnalyticsQuery;
@@ -0,0 +1,80 @@
1
+ /**
2
+ * The real {@link AnalyticsQuery}: DuckDB with the S3 Tables catalog attached
3
+ * read-only. Everything about the session - the credential secret, the attach,
4
+ * the quoting and the translation of vendor errors - belongs to
5
+ * `duckdb-session.ts`, which the write adapter beside this one shares; what
6
+ * lives here is only what reading a named query adds to it.
7
+ *
8
+ * **Read-only is a property of the connection, not of this module's shape.**
9
+ * The session is built with `readOnly: true`, so the `ATTACH` carries
10
+ * `READ_ONLY` and DuckDB itself refuses a write on it. `AnalyticsQuery.run` is
11
+ * the whole surface this adapter returns, so there is no statement to supply
12
+ * and no write path to reach even before that.
13
+ */
14
+ import { PAGE_VIEWS_RELATION, prepareQuery, } from '../queries.js';
15
+ import { createDuckDbSession, } from './duckdb-session.js';
16
+ /**
17
+ * {@link PAGE_VIEWS_RELATION} wherever a statement names it as a whole word.
18
+ * The word boundaries matter: `daily_page_views` must not match, and a bound
19
+ * relation contains the literal `page_views` inside its own quotes, which a
20
+ * replacement callback (rather than a replacement string) leaves alone.
21
+ */
22
+ const PAGE_VIEWS_RELATION_PATTERN = new RegExp(String.raw `\b${PAGE_VIEWS_RELATION}\b`, 'g');
23
+ /**
24
+ * The statement a prepared query actually runs as: its definition's SQL with
25
+ * the fixed relation name rewritten to `relation`. Raises when a definition
26
+ * names no relation at all, because a statement that reads nothing would
27
+ * otherwise return an empty result that looks like "no traffic that week".
28
+ */
29
+ export function bindPageViewsRelation(prepared, relation) {
30
+ let bindings = 0;
31
+ const sql = prepared.sql.replaceAll(PAGE_VIEWS_RELATION_PATTERN, () => {
32
+ bindings += 1;
33
+ return relation;
34
+ });
35
+ if (bindings === 0) {
36
+ throw new Error(`analytics query "${prepared.name}" names no ${PAGE_VIEWS_RELATION} relation to bind, so it would read nothing`);
37
+ }
38
+ return sql;
39
+ }
40
+ /**
41
+ * Build the DuckDB-backed {@link AnalyticsQuery}. Returns the port and nothing
42
+ * wider: `run(name, params)` is the whole surface, so there is no statement to
43
+ * supply and no write path to reach. Construct it at the plugin's composition
44
+ * root - the dashboard, status and backfill commands - and hand every domain
45
+ * module the port.
46
+ *
47
+ * The connection is opened lazily on the first query and then reused, so
48
+ * building the adapter touches neither the network nor the native library, and
49
+ * the dashboard binds its port before AWS is ever consulted.
50
+ */
51
+ export function createDuckDbAnalyticsQuery(opts) {
52
+ const session = createDuckDbSession({
53
+ ctx: opts.ctx,
54
+ credentials: opts.credentials,
55
+ readOnly: true,
56
+ connect: opts.connect,
57
+ });
58
+ function contextualise(name, err) {
59
+ return new Error(`analytics query "${name}" against ${session.attachTarget} failed while ${session.detail(err, 'running the query')}`);
60
+ }
61
+ return {
62
+ async run(name, params) {
63
+ const prepared = prepareQuery(name, params, session.config);
64
+ const sql = bindPageViewsRelation(prepared, session.relation);
65
+ let connection;
66
+ try {
67
+ connection = await session.open();
68
+ }
69
+ catch (err) {
70
+ throw contextualise(prepared.name, err);
71
+ }
72
+ try {
73
+ return await session.step('executing the statement', () => connection.run(sql, prepared.bindings));
74
+ }
75
+ catch (err) {
76
+ throw contextualise(prepared.name, err);
77
+ }
78
+ },
79
+ };
80
+ }