openship 0.4.8 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +27 -1
- package/dist/bare-FSUAM5V5.js +23 -0
- package/dist/{bowser-O3AEA6AN.js → bowser-HNOTPGJO.js} +2 -2
- package/dist/{chunk-5MZ624VX.js → chunk-2GWYYHOR.js} +1 -1
- package/dist/chunk-3FYIXVX5.js +8051 -0
- package/dist/{chunk-ILFETZOM.js → chunk-3LDWWY7U.js} +21 -1
- package/dist/{chunk-W73X43R7.js → chunk-3NQQXMDR.js} +3 -3
- package/dist/{chunk-OWED2GRM.js → chunk-42WQABXM.js} +2 -2
- package/dist/{chunk-TYVME3XD.js → chunk-6FH4AXVY.js} +20 -3
- package/dist/{chunk-OXWDY22G.js → chunk-6ODUBKUB.js} +3 -3
- package/dist/{chunk-DYPH4HCI.js → chunk-AO252K7Z.js} +1 -1
- package/dist/{chunk-XPPZ7B2J.js → chunk-B3CE2OMR.js} +14 -12
- package/dist/{chunk-IXCHURVK.js → chunk-DOD2BXHQ.js} +4 -4
- package/dist/{chunk-VWPEKOH6.js → chunk-DRK3MEDS.js} +3 -3
- package/dist/{chunk-V6AN6XLZ.js → chunk-EMXI4UCU.js} +125 -47
- package/dist/{chunk-GKHUSZCD.js → chunk-EP2TNALB.js} +2 -2
- package/dist/{chunk-SLTYVHU7.js → chunk-GUQ3QCJ7.js} +942 -519
- package/dist/chunk-I3DPMTSF.js +29 -0
- package/dist/{chunk-L2VOGB2X.js → chunk-IW2UKBVF.js} +3 -3
- package/dist/{chunk-VDMQJHPI.js → chunk-IXM5245F.js} +228 -13
- package/dist/{chunk-MZ2WEHHY.js → chunk-JIMUYEKN.js} +4 -4
- package/dist/chunk-OGY63DGD.js +705 -0
- package/dist/{chunk-O6ZZF52V.js → chunk-ONKEYSMY.js} +2 -2
- package/dist/{chunk-6APULWSU.js → chunk-OTLSJYAC.js} +1744 -118
- package/dist/{chunk-EI575GHN.js → chunk-PDW4L3NV.js} +3 -3
- package/dist/{chunk-5NYUCONN.js → chunk-PSZSKEPA.js} +33 -51
- package/dist/{chunk-KI2EY3WB.js → chunk-PXKCZB2I.js} +1 -1
- package/dist/{chunk-FXTBHTWY.js → chunk-QEH4QUBX.js} +649 -85
- package/dist/{chunk-6APIPIJY.js → chunk-QWHBRY2K.js} +5 -5
- package/dist/{chunk-6B5YCFFV.js → chunk-RVER2EOO.js} +5 -5
- package/dist/{chunk-Q2CIRAOK.js → chunk-RXQBD4ZY.js} +1 -1
- package/dist/{chunk-TGIINPVM.js → chunk-TWJCLD4W.js} +60 -10
- package/dist/{chunk-4Z5QO7N6.js → chunk-VDDMADOF.js} +3 -3
- package/dist/{chunk-BBHX7T4R.js → chunk-VE2YOK4I.js} +991 -254
- package/dist/{chunk-2D3I5G6P.js → chunk-VZ2ZHPS2.js} +73 -43
- package/dist/{chunk-AVIB7CT7.js → chunk-WNWQBB7F.js} +2 -2
- package/dist/{chunk-FRNAR33O.js → chunk-XZYFZ6A5.js} +3 -3
- package/dist/{chunk-RPHRPFEH.js → chunk-ZCFVHA25.js} +1 -1
- package/dist/{chunk-FPRHYBY2.js → chunk-ZFJXPCQP.js} +1 -1
- package/dist/{chunk-JYGD5WSN.js → chunk-ZWLDLTJP.js} +1 -1
- package/dist/cloud-E5M6VMCG.js +31 -0
- package/dist/{cloud-7QYOGF7F.js → cloud-KRTZ2IL5.js} +3 -3
- package/dist/{detect-AZLOHUFX.js → detect-ALF4GXRX.js} +7 -8
- package/dist/{dist-IQJ6H6UZ.js → dist-BM6NPWIH.js} +3 -3
- package/dist/{dist-es-X3B5QYKB.js → dist-es-2YRDXMUU.js} +6 -6
- package/dist/{dist-es-OSNJBLXB.js → dist-es-CCVZEFRZ.js} +5 -5
- package/dist/{dist-es-VNKHQVIB.js → dist-es-IOH6G4GM.js} +4 -4
- package/dist/{dist-es-RYHIPOTN.js → dist-es-JGBPABJI.js} +14 -14
- package/dist/{dist-es-JSTDUS7Q.js → dist-es-KWSMDPIP.js} +6 -6
- package/dist/{dist-es-5DY7K32L.js → dist-es-UYDDOVZ3.js} +6 -6
- package/dist/{dist-es-WV4AMOII.js → dist-es-XUOR4NQN.js} +8 -8
- package/dist/docker-QECQJ5P4.js +32 -0
- package/dist/docker-edge-executor-62N64MGS.js +17 -0
- package/dist/edge-import-CQWHN2Q3.js +48 -0
- package/dist/ensure-container-edge-MJVXJQG5.js +32 -0
- package/dist/{event-streams-EHRAKL75.js → event-streams-TR4GAC53.js} +4 -4
- package/dist/{executor-XJ6RP4Q2.js → executor-JELARVKC.js} +8 -8
- package/dist/index.js +1209 -560
- package/dist/{loadSso-46FQRIZS.js → loadSso-UGF4JMRK.js} +10 -10
- package/dist/{local-executor-V7U5VOZW.js → local-executor-WOFZD72X.js} +4 -4
- package/dist/nginx-AXGNWRKZ.js +31 -0
- package/dist/node-entry.js +4 -0
- package/dist/{noop-FMMFG2EZ.js → noop-NJ72F46T.js} +3 -3
- package/dist/openresty-lua-IE4P5YQX.js +81 -0
- package/dist/server/index.js +116248 -94749
- package/dist/server/lua/geo_country.lua +31 -6
- package/dist/server/lua/maxminddb.lua +403 -0
- package/dist/server/lua/mgmt_api.lua +431 -68
- package/dist/server/lua/pipe_log.lua +10 -4
- package/dist/server/lua/site_logger.lua +220 -12
- package/dist/server/migrations/0073_project_volumes_object_storage.sql +3 -0
- package/dist/server/migrations/0074_project_compose_path.sql +1 -0
- package/dist/server/migrations/0075_service_command_argv.sql +1 -0
- package/dist/server/migrations/0076_rollback_retention.sql +17 -0
- package/dist/server/migrations/0077_domain_redirect.sql +13 -0
- package/dist/server/migrations/0078_grant_source_scope.sql +30 -0
- package/dist/server/migrations/0079_project_readiness.sql +18 -0
- package/dist/server/migrations/0080_deployment_error_classification.sql +26 -0
- package/dist/server/migrations/0081_project_readiness_repair.sql +34 -0
- package/dist/server/migrations/0082_edge_target_verification.sql +39 -0
- package/dist/server/migrations/0083_service_incident.sql +84 -0
- package/dist/server/migrations/0084_analytics_daily_rollup.sql +72 -0
- package/dist/server/migrations/0085_resource_usage.sql +68 -0
- package/dist/server/migrations/0086_remote_infra_updates.sql +41 -0
- package/dist/server/migrations/0087_server_container_version_columns.sql +17 -0
- package/dist/server/migrations/0088_audit_source_and_settings.sql +27 -0
- package/dist/server/migrations/0089_auto_scan_infra.sql +7 -0
- package/dist/server/migrations/0090_project_collect_paths.sql +18 -0
- package/dist/server/migrations/0091_project_server_id.sql +26 -0
- package/dist/server/migrations/0092_project_internal_alias.sql +13 -0
- package/dist/server/migrations/0093_service_incident_org_index.sql +13 -0
- package/dist/server/migrations/0094_backup_restore_meta.sql +15 -0
- package/dist/server/migrations/0095_backup_restore_cancel.sql +19 -0
- package/dist/server/migrations/0096_backup_policy_retention_defaults.sql +20 -0
- package/dist/server/migrations/0097_update_status_upstream_only.sql +35 -0
- package/dist/server/migrations/meta/_journal.json +175 -0
- package/dist/setup-FA4PW4NU.js +21 -0
- package/dist/{signin-ITOPKKUB.js → signin-3M3UPB6K.js} +10 -10
- package/dist/{sso-oidc-NON2PD7Y.js → sso-oidc-3VDIYOEJ.js} +10 -10
- package/dist/{sts-HHFFSAY2.js → sts-B3WIYLXR.js} +9 -9
- package/package.json +6 -3
- package/dist/bare-JY5Y5ELU.js +0 -23
- package/dist/chunk-SJIFLLX4.js +0 -2931
- package/dist/chunk-VFINNTUL.js +0 -362
- package/dist/cloud-KYWY3RP4.js +0 -30
- package/dist/docker-LODK5Y2S.js +0 -16
- package/dist/docker-edge-executor-VHW3JPJS.js +0 -13
- package/dist/edge-import-ABUDKKJI.js +0 -45
- package/dist/ensure-container-edge-SW6HENSG.js +0 -31
- package/dist/nginx-3L4SIBJA.js +0 -24
- package/dist/openresty-lua-4WLGDSER.js +0 -42
- package/dist/setup-NU6T2PDU.js +0 -20
|
@@ -8,17 +8,31 @@
|
|
|
8
8
|
-- s:{domain}:{epoch_min}:i bandwidth in bytes (TTL 24h)
|
|
9
9
|
-- s:{domain}:{epoch_min}:o bandwidth out bytes (TTL 24h)
|
|
10
10
|
-- s:{domain}:{epoch_min}:t response time sum (seconds) (TTL 24h)
|
|
11
|
-
-- s:{domain}:{epoch_min}:u
|
|
12
|
-
-- g:{domain}:{YYYYMMDD}:{CC} country hit count (TTL 48h)
|
|
11
|
+
-- s:{domain}:{epoch_min}:u page (non-static) requests (TTL 24h)
|
|
13
12
|
-- c:{domain}:{epoch_min}:{CC} country per minute (TTL 24h)
|
|
14
13
|
-- t:{domain}:r / :i / :o lifetime totals (no TTL)
|
|
15
14
|
-- d:{domain} domain index marker (no TTL)
|
|
16
15
|
--
|
|
16
|
+
-- Daily rollup, all under one `g:{domain}:{YYYYMMDD}:` prefix so mgmt_api's
|
|
17
|
+
-- /analytics/geo serves the whole day in a SINGLE dict scan:
|
|
18
|
+
-- g:{domain}:{YYYYMMDD}:{CC} country hit count (TTL 48h)
|
|
19
|
+
-- g:{domain}:{YYYYMMDD}:v distinct visitors (TTL 48h)
|
|
20
|
+
-- g:{domain}:{YYYYMMDD}:p:{path} path hit count (TTL 48h)
|
|
21
|
+
-- g:{domain}:{YYYYMMDD}:s:{code} status-code count (TTL 48h)
|
|
22
|
+
--
|
|
23
|
+
-- Shared dict key schema (visitors zone):
|
|
24
|
+
-- vsalt:{YYYYMMDD} per-day hash salt (TTL 48h)
|
|
25
|
+
-- vd:{domain}:{YYYYMMDD}:{hash} distinct-visitor marker (TTL 48h)
|
|
26
|
+
--
|
|
17
27
|
-- Shared dict key schema (request_data zone):
|
|
18
28
|
-- rlog:{domain}:seq monotonic write pointer
|
|
19
29
|
-- rlog:{domain}:{slot} JSON entry (ring buf) (TTL 1h)
|
|
20
30
|
-- log_pipe:sub:{domain} live subscriber flag (TTL 30s)
|
|
21
31
|
-- log_pipe:q:{domain} live log queue entries
|
|
32
|
+
--
|
|
33
|
+
-- NOTE on `:u` — it counts non-static REQUESTS, not people. It was surfaced as
|
|
34
|
+
-- "unique IPs" all the way to the dashboard, which it never was. Real distinct
|
|
35
|
+
-- visitors are the `vd:`/`:v` pair below.
|
|
22
36
|
|
|
23
37
|
local cjson = require "cjson.safe"
|
|
24
38
|
|
|
@@ -35,6 +49,15 @@ if not pipe_ok then pipe_log = nil end
|
|
|
35
49
|
|
|
36
50
|
local analytics = ngx.shared.analytics
|
|
37
51
|
local request_data = ngx.shared.request_data
|
|
52
|
+
-- Per-host edge config, pushed by the API (see mgmt_api's /analytics/config).
|
|
53
|
+
--
|
|
54
|
+
-- The `rules` zone, NOT `analytics`: this holds a handful of small config values and is
|
|
55
|
+
-- never under eviction pressure, whereas the analytics zone is a 256 MB churn of counters.
|
|
56
|
+
-- An evicted flag there would silently stop path collection with nothing to explain why.
|
|
57
|
+
local rules_dict = ngx.shared.rules
|
|
58
|
+
-- Optional: a box whose nginx.conf predates this zone keeps working, just without
|
|
59
|
+
-- distinct-visitor counts. Never assume a dict exists.
|
|
60
|
+
local visitors = ngx.shared.visitors
|
|
38
61
|
if not analytics or not request_data then return end
|
|
39
62
|
|
|
40
63
|
-- ── Helpers ──────────────────────────────────────────────────────────────────
|
|
@@ -89,11 +112,71 @@ local function is_static_asset(u)
|
|
|
89
112
|
return false
|
|
90
113
|
end
|
|
91
114
|
|
|
115
|
+
-- Collapse a request URI into a low-cardinality bucket suitable as a dict key.
|
|
116
|
+
-- Without this, "top paths" is unbounded: every ?token=, every /orders/48219, and
|
|
117
|
+
-- every 404 a scanner probes mints a permanent key, and the analytics zone
|
|
118
|
+
-- LRU-evicts real counters to store junk.
|
|
119
|
+
local function normalize_path(u)
|
|
120
|
+
-- Query string carries ids/tokens and never identifies the route.
|
|
121
|
+
local path = u:match("^([^?#]*)") or u
|
|
122
|
+
if path == "" then return "/" end
|
|
123
|
+
-- Numeric and UUID/hex segments are ids, not routes: /orders/48219 and
|
|
124
|
+
-- /orders/48220 are one path.
|
|
125
|
+
path = path:gsub("/%d+", "/:id")
|
|
126
|
+
path = path:gsub("/%x%x%x%x%x%x%x%x%-?%x*%-?%x*%-?%x*%-?%x*", "/:id")
|
|
127
|
+
-- Bound the key itself — a long URI must not become a long key.
|
|
128
|
+
if #path > 120 then path = path:sub(1, 120) .. "…" end
|
|
129
|
+
return path
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
-- ── Enumerable sets, without a zone scan ─────────────────────────────────────
|
|
133
|
+
--
|
|
134
|
+
-- Reading these counters back used to need key DISCOVERY, and a shared dict offers exactly
|
|
135
|
+
-- one mechanism for that: `get_keys`, which walks EVERY key in the zone holding its lock.
|
|
136
|
+
-- Cost scaled with how much was in the 256 MB zone, not with what was being asked for, and
|
|
137
|
+
-- it ran on the mgmt path four different ways.
|
|
138
|
+
--
|
|
139
|
+
-- So every unbounded set now carries its own index: a counter plus numbered slots.
|
|
140
|
+
--
|
|
141
|
+
-- <counter> how many distinct members
|
|
142
|
+
-- <prefix><n> the n-th member
|
|
143
|
+
--
|
|
144
|
+
-- A reader reads the counter and then reads slots 1..n. Bounded by the answer.
|
|
145
|
+
--
|
|
146
|
+
-- The write only happens on a member's FIRST sighting, so steady state costs nothing: the
|
|
147
|
+
-- millionth request from Germany today takes the plain `incr` path and touches no index.
|
|
148
|
+
local function index_add(counter_key, slot_prefix, member, ttl)
|
|
149
|
+
local n = analytics:incr(counter_key, 1, 0, ttl)
|
|
150
|
+
if n then analytics:set(slot_prefix .. n, member, ttl) end
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
--- Add 1 to a counter, creating it (with TTL) and indexing it on first sighting.
|
|
154
|
+
--
|
|
155
|
+
-- `incr` with no `init` returns nil for an absent key, which doubles as the existence
|
|
156
|
+
-- probe — so the common case is ONE hash operation. TTL is set by the creating `safe_add`
|
|
157
|
+
-- rather than passed to `incr`: shdict rejects `init_ttl` without `init`, and that raises a
|
|
158
|
+
-- Lua error which aborts the whole log handler (it has silently zeroed every counter below
|
|
159
|
+
-- it before).
|
|
160
|
+
local function bump_indexed(key, ttl, counter_key, slot_prefix, member)
|
|
161
|
+
if analytics:incr(key, 1) then return end
|
|
162
|
+
if analytics:safe_add(key, 1, ttl) then
|
|
163
|
+
index_add(counter_key, slot_prefix, member, ttl)
|
|
164
|
+
else
|
|
165
|
+
-- Another worker created it between our probe and our add; our +1 still counts.
|
|
166
|
+
analytics:incr(key, 1)
|
|
167
|
+
end
|
|
168
|
+
end
|
|
169
|
+
|
|
92
170
|
local RING = 1000
|
|
93
171
|
local D24H = 86400
|
|
94
172
|
local D48H = 172800
|
|
95
173
|
local D1H = 3600
|
|
96
174
|
|
|
175
|
+
-- Distinct paths tracked per domain per day. Past this the tail folds into
|
|
176
|
+
-- "other" rather than growing the zone — a URL-fuzzing bot is capped at one
|
|
177
|
+
-- extra key, not thousands.
|
|
178
|
+
local PATH_CARDINALITY_CAP = 2000
|
|
179
|
+
|
|
97
180
|
-- ── Capture ──────────────────────────────────────────────────────────────────
|
|
98
181
|
|
|
99
182
|
local host = normalize(ngx.var.host)
|
|
@@ -109,6 +192,22 @@ local rt = tonumber(ngx.var.request_time) or 0 -- seconds (float)
|
|
|
109
192
|
local method = ngx.var.request_method or "GET"
|
|
110
193
|
local status = ngx.var.status or "0"
|
|
111
194
|
|
|
195
|
+
-- Client country, resolved ONCE per request.
|
|
196
|
+
--
|
|
197
|
+
-- Hoisted out of the geo block because THREE consumers need it — the daily/per-minute
|
|
198
|
+
-- rollups, the raw ring buffer behind /logs/recent, and the live SSE pipe. It used to be
|
|
199
|
+
-- looked up inside the rollup and again inside pipe_log, so the ring buffer had no
|
|
200
|
+
-- country at all: a request-log list showed flags on rows that arrived live and none on
|
|
201
|
+
-- rows backfilled from /logs/recent, for the same traffic.
|
|
202
|
+
--
|
|
203
|
+
-- pcall'd because a corrupt or partially-written mmdb makes the lookup raise, and an
|
|
204
|
+
-- error here would abort every counter below it (see the section header).
|
|
205
|
+
local cc = nil
|
|
206
|
+
if geo and geo.get_country_code then
|
|
207
|
+
local ok_cc, res = pcall(geo.get_country_code, ip)
|
|
208
|
+
if ok_cc and type(res) == "string" and res ~= "" then cc = res end
|
|
209
|
+
end
|
|
210
|
+
|
|
112
211
|
-- ── 1. Minute-bucket counters ────────────────────────────────────────────────
|
|
113
212
|
|
|
114
213
|
local minute = math.floor(ts / 60)
|
|
@@ -136,26 +235,135 @@ analytics:incr("t:" .. host .. ":r", 1, 0)
|
|
|
136
235
|
analytics:incr("t:" .. host .. ":i", req_len, 0)
|
|
137
236
|
analytics:incr("t:" .. host .. ":o", bytes, 0)
|
|
138
237
|
|
|
139
|
-
-- Domain index
|
|
140
|
-
|
|
238
|
+
-- Domain index. `safe_add` returning true means this is the first request this box has
|
|
239
|
+
-- ever seen for the host, which is exactly when the enumerable index needs appending —
|
|
240
|
+
-- so domain discovery no longer costs a zone scan either.
|
|
241
|
+
if analytics:safe_add("d:" .. host, 1) then
|
|
242
|
+
index_add("dn", "di:", host)
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
-- ── 3. Daily rollup: geo, visitors, paths, statuses ──────────────────────────
|
|
246
|
+
--
|
|
247
|
+
-- Each sub-block is pcall'd SEPARATELY, and that is load-bearing rather than
|
|
248
|
+
-- defensive habit. A raised error in log_by_lua aborts the remainder of the
|
|
249
|
+
-- handler, so one bad shdict call silently zeroes every counter written after it —
|
|
250
|
+
-- which is exactly what an `incr(..., nil, ttl)` here did: paths AND statuses read
|
|
251
|
+
-- as "no traffic" while requests and bandwidth looked perfect. Isolating them
|
|
252
|
+
-- means a future edit can lose at most its own metric.
|
|
141
253
|
|
|
142
|
-
|
|
254
|
+
local day = today()
|
|
255
|
+
local gpfx = "g:" .. host .. ":" .. day .. ":"
|
|
143
256
|
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
257
|
+
-- 3a. GeoIP. `cc` was resolved once above and is reused by the ring buffer and the
|
|
258
|
+
-- live pipe, so a request costs exactly one mmdb lookup no matter how many consumers
|
|
259
|
+
-- report its country.
|
|
260
|
+
if cc then
|
|
261
|
+
pcall(function()
|
|
147
262
|
-- Daily geo (for /analytics/geo endpoint)
|
|
148
|
-
|
|
149
|
-
-- Per-minute geo (for time-series country breakdown)
|
|
150
|
-
|
|
263
|
+
bump_indexed(gpfx .. cc, D48H, gpfx .. "cn", gpfx .. "ci:", cc)
|
|
264
|
+
-- Per-minute geo (for the time-series country breakdown). Indexed per MINUTE, so
|
|
265
|
+
-- the scraper reads only the minutes in its window.
|
|
266
|
+
local mpfx = "c:" .. host .. ":" .. minute .. ":"
|
|
267
|
+
bump_indexed(mpfx .. cc, D24H, mpfx .. "n", mpfx .. "i:", cc)
|
|
268
|
+
end)
|
|
269
|
+
end
|
|
270
|
+
|
|
271
|
+
-- 3b. Distinct visitors — a COUNT, never an identity.
|
|
272
|
+
--
|
|
273
|
+
-- The address is hashed with a salt that is generated on this box, rotates daily,
|
|
274
|
+
-- and is never written anywhere but this shared dict, so the marker keys are not
|
|
275
|
+
-- reversible to an IP and are worthless the next day. Only the resulting counter
|
|
276
|
+
-- is ever flushed to Postgres; no per-visitor row exists at any layer. That is
|
|
277
|
+
-- what lets us report real visitor numbers while still not collecting behavioural
|
|
278
|
+
-- analytics on anyone's end users.
|
|
279
|
+
--
|
|
280
|
+
-- Per DAY, not per minute: a per-minute set is ~1440x the keys for a number
|
|
281
|
+
-- nobody reads, and would evict the counters it exists to annotate.
|
|
282
|
+
pcall(function()
|
|
283
|
+
if visitors then
|
|
284
|
+
local salt_key = "vsalt:" .. day
|
|
285
|
+
local salt = visitors:get(salt_key)
|
|
286
|
+
if not salt then
|
|
287
|
+
-- First writer wins; a racing worker's safe_add fails and it re-reads. The
|
|
288
|
+
-- salt only has to be unpredictable and stable for the day.
|
|
289
|
+
salt = tostring(ngx.now()) .. tostring(math.random(1, 2147483647))
|
|
290
|
+
local ok = visitors:safe_add(salt_key, salt, D48H)
|
|
291
|
+
if not ok then salt = visitors:get(salt_key) or salt end
|
|
292
|
+
end
|
|
293
|
+
-- safe_add is the dedup: it succeeds ONLY on this visitor's first request
|
|
294
|
+
-- today, so the counter below increments exactly once per distinct visitor.
|
|
295
|
+
local marker = "vd:" .. host .. ":" .. day .. ":" .. ngx.crc32_long(salt .. ip)
|
|
296
|
+
if visitors:safe_add(marker, 1, D48H) then
|
|
297
|
+
analytics:incr(gpfx .. "v", 1, 0, D48H)
|
|
151
298
|
end
|
|
152
299
|
end
|
|
300
|
+
end)
|
|
301
|
+
|
|
302
|
+
-- 3c. Top paths — OPT-IN, non-static only, normalized and cardinality-capped.
|
|
303
|
+
--
|
|
304
|
+
-- Off unless the project turned it on, because this one block is the most expensive thing
|
|
305
|
+
-- the log handler does. Measured on the shipped edge image, per request:
|
|
306
|
+
--
|
|
307
|
+
-- paths (whole block) 1.72 us <- 57% of the counter path
|
|
308
|
+
-- of which string work 1.38 us normalize_path + is_static_asset
|
|
309
|
+
-- minute buckets (5 incr) 0.61 us
|
|
310
|
+
-- country (2 incr) 0.24 us
|
|
311
|
+
-- status (1 incr) 0.14 us
|
|
312
|
+
-- this flag read (1 get) 0.07 us
|
|
313
|
+
--
|
|
314
|
+
-- So the gate costs 0.07 us and saves 1.72 us — the counter path drops from ~3.0 us to
|
|
315
|
+
-- ~1.35 us with paths off. It is also the highest-cardinality dimension (up to
|
|
316
|
+
-- PATH_CARDINALITY_CAP keys per domain per day, against ~200 countries and a few dozen
|
|
317
|
+
-- statuses) and the largest column in the daily rollup.
|
|
318
|
+
--
|
|
319
|
+
-- Absent flag = off. A box whose dict was just restarted therefore collects nothing until
|
|
320
|
+
-- the API re-pushes, which is the safe direction: no data beats data nobody asked to pay
|
|
321
|
+
-- for. The DB is the source of truth and re-pushes on every route apply.
|
|
322
|
+
local collect_paths = rules_dict and rules_dict:get("cfg:" .. host .. ":paths")
|
|
323
|
+
|
|
324
|
+
pcall(function()
|
|
325
|
+
if collect_paths and not is_static_asset(uri) then
|
|
326
|
+
-- Hoisted: normalize_path is the single most expensive call in this handler (~1.4 us
|
|
327
|
+
-- of string work), so it runs exactly once even though both the counter key and the
|
|
328
|
+
-- index slot need its result.
|
|
329
|
+
local norm = normalize_path(uri)
|
|
330
|
+
local path_key = gpfx .. "p:" .. norm
|
|
331
|
+
-- incr-then-add: the hot path is ONE hash hit on an existing key. `incr` with
|
|
332
|
+
-- no init returns nil for an absent key, which doubles as the existence probe,
|
|
333
|
+
-- so only a path unseen today reaches the cap check below.
|
|
334
|
+
--
|
|
335
|
+
-- Deliberately no init_ttl here: shdict rejects init_ttl unless init is also
|
|
336
|
+
-- given ("'init_ttl' must be used with 'init'"), and that raises a Lua error
|
|
337
|
+
-- which aborts the whole log handler — taking the status counters below down
|
|
338
|
+
-- with it. The TTL is set by the safe_add on the creating request instead.
|
|
339
|
+
if not analytics:incr(path_key, 1) then
|
|
340
|
+
local distinct = analytics:incr(gpfx .. "pn", 1, 0, D48H) or 0
|
|
341
|
+
if distinct > PATH_CARDINALITY_CAP then
|
|
342
|
+
analytics:incr(gpfx .. "p:other", 1, 0, D48H)
|
|
343
|
+
else
|
|
344
|
+
-- `pn` was already the distinct-path counter, so the slots just give it an
|
|
345
|
+
-- enumerable side. Stored WITHOUT the "p:" prefix; the reader adds it back.
|
|
346
|
+
if analytics:safe_add(path_key, 1, D48H) then
|
|
347
|
+
analytics:set(gpfx .. "pi:" .. distinct, norm, D48H)
|
|
348
|
+
end
|
|
349
|
+
end
|
|
350
|
+
end
|
|
351
|
+
end
|
|
352
|
+
end)
|
|
353
|
+
|
|
354
|
+
-- 3d. Status-code mix. Bounded by construction (a few dozen codes).
|
|
355
|
+
pcall(function()
|
|
356
|
+
bump_indexed(gpfx .. "s:" .. status, D48H, gpfx .. "sn", gpfx .. "si:", status)
|
|
357
|
+
end)
|
|
153
358
|
|
|
154
359
|
-- ── 4. Raw request ring buffer ──────────────────────────────────────────────
|
|
155
360
|
|
|
156
361
|
pcall(function()
|
|
157
362
|
local ok_j, j = pcall(cjson.encode, {
|
|
158
363
|
ip = ip,
|
|
364
|
+
-- nil is simply omitted by cjson, so a box with no GeoIP writes the same shape
|
|
365
|
+
-- minus this key rather than a null the reader has to special-case.
|
|
366
|
+
country = cc,
|
|
159
367
|
ts = ts,
|
|
160
368
|
method = method,
|
|
161
369
|
status = status,
|
|
@@ -178,5 +386,5 @@ end)
|
|
|
178
386
|
if pipe_log and pipe_log.pipe_request_log
|
|
179
387
|
and request_data:get("log_pipe:sub:" .. host) then
|
|
180
388
|
ngx.timer.at(0, pipe_log.pipe_request_log,
|
|
181
|
-
host, ip, ts, ua, uri, req_len, bytes, rt, method, status)
|
|
389
|
+
host, ip, ts, ua, uri, req_len, bytes, rt, method, status, cc)
|
|
182
390
|
end
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
ALTER TABLE "project" ADD COLUMN IF NOT EXISTS "compose_path" text;
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
ALTER TABLE "service" ADD COLUMN IF NOT EXISTS "command_argv" jsonb;
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
-- Rollback retention: auto-sized window inputs on the project, and the removal
|
|
2
|
+
-- of a per-service artifact flag that was never written or read.
|
|
3
|
+
--
|
|
4
|
+
-- `rollback_window` keeps its existing meaning with one addition: NULL now means
|
|
5
|
+
-- "auto" — resolve to `rollback_window_computed` (measured from the deploy host's
|
|
6
|
+
-- free disk and this project's average snapshot size), falling back to
|
|
7
|
+
-- instance_settings.default_rollback_window when nothing was measured yet. A
|
|
8
|
+
-- non-null `rollback_window` is still an explicit operator override.
|
|
9
|
+
ALTER TABLE "project" ADD COLUMN IF NOT EXISTS "snapshot_size_bytes" bigint;
|
|
10
|
+
--> statement-breakpoint
|
|
11
|
+
ALTER TABLE "project" ADD COLUMN IF NOT EXISTS "rollback_window_computed" integer;
|
|
12
|
+
--> statement-breakpoint
|
|
13
|
+
ALTER TABLE "project" ADD COLUMN IF NOT EXISTS "capacity_measured_at" timestamp;
|
|
14
|
+
--> statement-breakpoint
|
|
15
|
+
-- Dead since it was introduced: the deployment-level `artifact_retained_at` is
|
|
16
|
+
-- the only retention flag any code path writes or reads.
|
|
17
|
+
ALTER TABLE "service_deployment" DROP COLUMN IF EXISTS "artifact_retained_at";
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
-- Per-domain canonical redirect: this hostname answers with a redirect to another
|
|
2
|
+
-- of the project's hostnames instead of serving the app.
|
|
3
|
+
--
|
|
4
|
+
-- Lives on `domain` rather than in a routing blob because the domain table IS the
|
|
5
|
+
-- store for a project's public endpoints (deriveProjectRouteState builds
|
|
6
|
+
-- publicEndpoints FROM these rows), and a redirect is a property of one hostname.
|
|
7
|
+
--
|
|
8
|
+
-- `redirect_to` NULL = serve the app (every existing row, unchanged).
|
|
9
|
+
-- `redirect_status` NULL = 301. Kept nullable rather than DEFAULT 301 so "no
|
|
10
|
+
-- redirect" is one state, not a status attached to a row that doesn't redirect.
|
|
11
|
+
ALTER TABLE "domain" ADD COLUMN IF NOT EXISTS "redirect_to" text;
|
|
12
|
+
--> statement-breakpoint
|
|
13
|
+
ALTER TABLE "domain" ADD COLUMN IF NOT EXISTS "redirect_status" integer;
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
-- Per-repo source access scope on every grant-bearing table.
|
|
2
|
+
--
|
|
3
|
+
-- A grant already carries a VERB in `permissions_json` (read / write / admin).
|
|
4
|
+
-- This column adds the SURFACE: may the holder read file CONTENT, and under which
|
|
5
|
+
-- paths. The two do not overlap, and the absence of a scope is meaningful:
|
|
6
|
+
--
|
|
7
|
+
-- read + scope_json NULL -> metadata only (branches,
|
|
8
|
+
-- detect, deploy). THE DEFAULT.
|
|
9
|
+
-- read + {"v":1,"read":{"paths":["src/**"]}} -> file content under src only
|
|
10
|
+
-- write + {"v":1,"write":{"paths":[...]}} -> content writes under those paths
|
|
11
|
+
--
|
|
12
|
+
-- So a repo grant no longer implies the right to crawl the repository. NULL is the
|
|
13
|
+
-- restrictive state, which is why this is nullable with no default and no backfill:
|
|
14
|
+
-- every existing row correctly reads as metadata-only.
|
|
15
|
+
--
|
|
16
|
+
-- Encoded as JSON text (not jsonb, not text[]) to match `permissions_json` — the
|
|
17
|
+
-- column types have to behave identically across PGlite and Postgres without
|
|
18
|
+
-- driver-specific casting, and the repos already own (de)serialisation.
|
|
19
|
+
--
|
|
20
|
+
-- Shape + matching semantics: packages/core/src/source-access.ts. Malformed JSON,
|
|
21
|
+
-- an unknown "v", or an unparseable pattern all resolve to NO access rather than
|
|
22
|
+
-- unrestricted access.
|
|
23
|
+
ALTER TABLE "resource_grant" ADD COLUMN IF NOT EXISTS "scope_json" text;
|
|
24
|
+
--> statement-breakpoint
|
|
25
|
+
ALTER TABLE "personal_access_token_grant" ADD COLUMN IF NOT EXISTS "scope_json" text;
|
|
26
|
+
--> statement-breakpoint
|
|
27
|
+
-- Pending invite grants carry the scope through acceptance; without it, a
|
|
28
|
+
-- restricted invitee would materialise into a metadata-only grant and silently
|
|
29
|
+
-- lose the content access the inviter picked.
|
|
30
|
+
ALTER TABLE "invitation_pending_grant" ADD COLUMN IF NOT EXISTS "scope_json" text;
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
-- Deploy-time readiness gate, per project.
|
|
2
|
+
--
|
|
3
|
+
-- Named `readiness`, NOT `health_check`, on purpose: a compose service already
|
|
4
|
+
-- carries a `healthcheck` (the Docker HEALTHCHECK directive, stored in
|
|
5
|
+
-- `service.advanced`) which the daemon runs and which has nothing to do with
|
|
6
|
+
-- whether a DEPLOY is allowed to finish. Two fields one capital letter apart in
|
|
7
|
+
-- the same openship.json would be a footgun.
|
|
8
|
+
--
|
|
9
|
+
-- Nullable with NO default, and no backfill: NULL means "off", so every existing
|
|
10
|
+
-- project moves to the new off-by-default behaviour. Previously the 15s
|
|
11
|
+
-- stabilization watch and the 45s TCP probe were hardcoded on every deploy and
|
|
12
|
+
-- could fail (and then force-remove) a container that was running fine. Opting
|
|
13
|
+
-- in is now explicit.
|
|
14
|
+
--
|
|
15
|
+
-- Shape (all fields optional, all defaults off) — see OpenshipReadiness:
|
|
16
|
+
-- { enabled, path, port, timeoutSeconds,
|
|
17
|
+
-- stabilization, stabilizationSeconds, onFailure: "warn" | "fail" }
|
|
18
|
+
ALTER TABLE "project" ADD COLUMN IF NOT EXISTS "readiness" jsonb;
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
-- Durable failure classification for a deployment.
|
|
2
|
+
--
|
|
3
|
+
-- The pipeline already classifies failures precisely: a port conflict throws
|
|
4
|
+
-- DeployError(msg, "PORT_IN_USE", {port, pid, command, systemdUnit,
|
|
5
|
+
-- isManagedDeployment, …}) and that reaches onFailure intact. It was then split,
|
|
6
|
+
-- and only half of it was kept: `error_message` went to this table while
|
|
7
|
+
-- `errorCode`/`errorDetails` went ONLY to the in-memory SSE session. Once that
|
|
8
|
+
-- session was evicted (or the API restarted) all that survived a port conflict
|
|
9
|
+
-- was an English sentence, so nothing could offer the operator — or an MCP agent
|
|
10
|
+
-- — a way forward.
|
|
11
|
+
--
|
|
12
|
+
-- The one attempt to recover the code was inverted: build-status re-derived it
|
|
13
|
+
-- with error_message.includes("PORT_IN_USE"), and none of the coded messages
|
|
14
|
+
-- contain that token (they read "Port 3000 is already in use by …"). So the
|
|
15
|
+
-- failures we DID classify lost their code and the ones we didn't got it guessed.
|
|
16
|
+
--
|
|
17
|
+
-- `error_code` is free text with no check constraint, matching `status` on this
|
|
18
|
+
-- table: codes come from DeployError call sites across packages/adapters and must
|
|
19
|
+
-- be extendable without a migration. `error_details` is the DeployError details
|
|
20
|
+
-- bag verbatim (never contains secrets — it is process/port metadata).
|
|
21
|
+
--
|
|
22
|
+
-- Both nullable with no backfill: NULL means "not classified", which is exactly
|
|
23
|
+
-- true of every pre-existing row.
|
|
24
|
+
ALTER TABLE "deployment" ADD COLUMN IF NOT EXISTS "error_code" text;
|
|
25
|
+
--> statement-breakpoint
|
|
26
|
+
ALTER TABLE "deployment" ADD COLUMN IF NOT EXISTS "error_details" jsonb;
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
-- Repair: create `project.readiness` on a database that skipped 0079.
|
|
2
|
+
--
|
|
3
|
+
-- WHY THIS EXISTS (and why it is not just "0079 again"):
|
|
4
|
+
--
|
|
5
|
+
-- 0079 was originally `0079_project_health_check` (adding `project.health_check`)
|
|
6
|
+
-- and was replaced in place by `0079_project_readiness` (adding
|
|
7
|
+
-- `project.readiness`) before either shipped. Both journal entries carry the SAME
|
|
8
|
+
-- `when` — 1785882107324 — because the second reused the first's slot.
|
|
9
|
+
--
|
|
10
|
+
-- drizzle's migrator reads only the single most recent applied row and applies a
|
|
11
|
+
-- migration when `lastDbMigration.created_at < migration.folderMillis`
|
|
12
|
+
-- (pg-core/dialect.js). That comparison is STRICT, so on any database that had
|
|
13
|
+
-- already run the old 0079, the new one is equal, not greater — it is skipped
|
|
14
|
+
-- without a word. 0080 carried a later stamp and applied normally, which is why
|
|
15
|
+
-- such a database ends up with 0080's columns, an orphan `health_check`, and no
|
|
16
|
+
-- `readiness` at all. Every query selecting the project row then dies with
|
|
17
|
+
-- `column "readiness" does not exist` (42703).
|
|
18
|
+
--
|
|
19
|
+
-- Bumping 0079's `when` would NOT fix it: the max applied stamp is now 0080's, so
|
|
20
|
+
-- anything ordered before 0080 still compares as already-applied. The repair has
|
|
21
|
+
-- to sort AFTER 0080, which is what this file is.
|
|
22
|
+
--
|
|
23
|
+
-- Idempotent on purpose. A fresh database creates the column at 0079 and this is
|
|
24
|
+
-- a no-op; a database that skipped 0079 creates it here. Either way the end state
|
|
25
|
+
-- is identical, so this is safe to leave in history permanently rather than being
|
|
26
|
+
-- a one-off someone has to remember to delete.
|
|
27
|
+
ALTER TABLE "project" ADD COLUMN IF NOT EXISTS "readiness" jsonb;--> statement-breakpoint
|
|
28
|
+
|
|
29
|
+
-- Drop the column the replaced migration left behind. `health_check` was only
|
|
30
|
+
-- ever created by 0079_project_health_check, which never shipped in a release and
|
|
31
|
+
-- is referenced by no schema, query, or type in the codebase — so on the
|
|
32
|
+
-- databases that have it, it is dead weight with a confusingly similar name to
|
|
33
|
+
-- the compose service `healthcheck`. Guarded, so this is a no-op everywhere else.
|
|
34
|
+
ALTER TABLE "project" DROP COLUMN IF EXISTS "health_check";
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
-- Proof that a routing TARGET is ours, as Openship Cloud's shared edge now requires
|
|
2
|
+
-- before forwarding `<slug>.opsh.io` to it.
|
|
3
|
+
--
|
|
4
|
+
-- Keyed on (organization_id, target) because that is what the upstream keys it on:
|
|
5
|
+
-- one target serves every free domain on that box, so this is target-level state,
|
|
6
|
+
-- not domain-level. One install holds several rows — its own address plus one per
|
|
7
|
+
-- remote server it deploys to.
|
|
8
|
+
--
|
|
9
|
+
-- The token is persisted HERE, not only on the edge, because the upstream's list
|
|
10
|
+
-- endpoint returns id/target/status/expiry but NOT the token. Verification lasts 90
|
|
11
|
+
-- days and the upstream re-probes THE SAME token within 7 days of expiry; a 404 then
|
|
12
|
+
-- expires it and the free domain stops resolving ~83 days after a green deploy with
|
|
13
|
+
-- nothing in our logs. So a token that lived only on a box's disk would be
|
|
14
|
+
-- unrecoverable after a rebuild, and rows here are never deleted when a free domain
|
|
15
|
+
-- is dropped.
|
|
16
|
+
--
|
|
17
|
+
-- `server_id` is ON DELETE SET NULL, not cascade: losing the server row must not
|
|
18
|
+
-- take the token with it.
|
|
19
|
+
CREATE TABLE IF NOT EXISTS "edge_target_verification" (
|
|
20
|
+
"id" text PRIMARY KEY NOT NULL,
|
|
21
|
+
"organization_id" text NOT NULL REFERENCES "organization"("id") ON DELETE CASCADE,
|
|
22
|
+
"target" text NOT NULL,
|
|
23
|
+
"host" text NOT NULL,
|
|
24
|
+
"server_id" text REFERENCES "servers"("id") ON DELETE SET NULL,
|
|
25
|
+
"verification_id" integer,
|
|
26
|
+
"token" text,
|
|
27
|
+
"retired_tokens" jsonb,
|
|
28
|
+
"challenge_path" text,
|
|
29
|
+
"status" text NOT NULL DEFAULT 'pending',
|
|
30
|
+
"validated_ip" text,
|
|
31
|
+
"expires_at" timestamp,
|
|
32
|
+
"last_checked_at" timestamp,
|
|
33
|
+
"last_error" text,
|
|
34
|
+
"created_at" timestamp NOT NULL DEFAULT now(),
|
|
35
|
+
"updated_at" timestamp NOT NULL DEFAULT now()
|
|
36
|
+
);
|
|
37
|
+
--> statement-breakpoint
|
|
38
|
+
CREATE UNIQUE INDEX IF NOT EXISTS "uq_edge_target_verification"
|
|
39
|
+
ON "edge_target_verification" ("organization_id", "target");
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
-- Continuous container health monitoring: the durable memory behind the alerts.
|
|
2
|
+
--
|
|
3
|
+
-- Openship knew a container had crashed at exactly ONE moment — the ~15s post-deploy
|
|
4
|
+
-- stabilization watch. After that nobody looked again, so an OOM-kill at 3am or a
|
|
5
|
+
-- Postgres that starts bouncing after a host reboot reached the operator only when a
|
|
6
|
+
-- user complained. The health watch closes that, but a poller that alerts on "the
|
|
7
|
+
-- container isn't running" gets muted within a day: a redeploy recreating containers,
|
|
8
|
+
-- an operator's `docker stop`, an unreachable host, and a crash loop all look exactly
|
|
9
|
+
-- like a failure at a point in time.
|
|
10
|
+
--
|
|
11
|
+
-- So the unit of alerting is an INCIDENT, not a state reading: opened once, escalated
|
|
12
|
+
-- only when it gets worse, resolved once with a downtime duration. This table is that
|
|
13
|
+
-- memory — it survives the control-plane restart that in-process state would not (a box
|
|
14
|
+
-- down for three days must not re-page on every API restart), and it doubles as the
|
|
15
|
+
-- project Health tab's history.
|
|
16
|
+
CREATE TABLE IF NOT EXISTS "service_incident" (
|
|
17
|
+
"id" text PRIMARY KEY NOT NULL,
|
|
18
|
+
"organization_id" text NOT NULL REFERENCES "organization"("id") ON DELETE CASCADE,
|
|
19
|
+
-- Null for SERVER-scoped incidents (an unreachable box gets ONE incident, not one
|
|
20
|
+
-- per service on it — that fan-out is the single loudest source of alert fatigue).
|
|
21
|
+
"project_id" text REFERENCES "project"("id") ON DELETE CASCADE,
|
|
22
|
+
-- The resolved service row, when there is one. SET NULL, not cascade: deleting a
|
|
23
|
+
-- service must not erase the record of the outage it had (`service_key` still
|
|
24
|
+
-- identifies it, and `service_name` is snapshotted below).
|
|
25
|
+
"service_id" text REFERENCES "service"("id") ON DELETE SET NULL,
|
|
26
|
+
-- Stable identity across a container RECREATE, which is the whole point: every
|
|
27
|
+
-- deploy mints a new container id, so keying incidents on container id would open a
|
|
28
|
+
-- fresh incident for the same workload after each redeploy. Service id when we
|
|
29
|
+
-- resolved one, else the container name (single-app deploys have no service row);
|
|
30
|
+
-- `server:<id>` for server-scoped rows.
|
|
31
|
+
"service_key" text NOT NULL,
|
|
32
|
+
-- Snapshot of the display name at incident time — the alert text must still read
|
|
33
|
+
-- correctly after a rename, and history rows must not silently change meaning.
|
|
34
|
+
"service_name" text NOT NULL,
|
|
35
|
+
"server_id" text REFERENCES "servers"("id") ON DELETE SET NULL,
|
|
36
|
+
"container_id" text,
|
|
37
|
+
-- down | crash_loop | unhealthy | server_unreachable. Ordered by severity in the
|
|
38
|
+
-- watcher: an incident escalates upward (and re-notifies), never downward.
|
|
39
|
+
"kind" text NOT NULL,
|
|
40
|
+
"status" text NOT NULL DEFAULT 'open', -- open | resolved
|
|
41
|
+
"reason" text,
|
|
42
|
+
"exit_code" integer,
|
|
43
|
+
"restart_count" integer NOT NULL DEFAULT 0,
|
|
44
|
+
"oom_killed" boolean NOT NULL DEFAULT false,
|
|
45
|
+
-- Consecutive ticks that agreed on this verdict. An incident is only opened (and
|
|
46
|
+
-- only notified) at >= 2, so a snapshot taken mid-restart or during a slow
|
|
47
|
+
-- healthcheck start never pages anyone.
|
|
48
|
+
"confirmations" integer NOT NULL DEFAULT 0,
|
|
49
|
+
-- Notification bookkeeping. Deliberately a count, not a flag: open, escalation and
|
|
50
|
+
-- resolve each notify exactly once, and nothing else ever does.
|
|
51
|
+
"notify_count" integer NOT NULL DEFAULT 0,
|
|
52
|
+
"notified_at" timestamp,
|
|
53
|
+
"log_excerpt" text,
|
|
54
|
+
"opened_at" timestamp NOT NULL DEFAULT now(),
|
|
55
|
+
"resolved_at" timestamp,
|
|
56
|
+
-- Last tick that actually OBSERVED this workload. Not advanced when the host is
|
|
57
|
+
-- unreachable — during a connectivity loss we know nothing, so nothing is claimed.
|
|
58
|
+
"last_seen_at" timestamp NOT NULL DEFAULT now(),
|
|
59
|
+
"created_at" timestamp NOT NULL DEFAULT now(),
|
|
60
|
+
"updated_at" timestamp NOT NULL DEFAULT now()
|
|
61
|
+
);
|
|
62
|
+
--> statement-breakpoint
|
|
63
|
+
-- One open incident per workload — the dedup primitive. As a partial unique index,
|
|
64
|
+
-- "we are already alerting about this" is a DB invariant rather than sweep
|
|
65
|
+
-- bookkeeping: two overlapping ticks (a manual Run now landing on a cron tick) cannot
|
|
66
|
+
-- both open one and double-notify.
|
|
67
|
+
CREATE UNIQUE INDEX IF NOT EXISTS "uq_service_incident_open"
|
|
68
|
+
ON "service_incident" ("project_id", "service_key")
|
|
69
|
+
WHERE "status" = 'open' AND "project_id" IS NOT NULL;
|
|
70
|
+
--> statement-breakpoint
|
|
71
|
+
-- Same invariant for server-scoped rows. Needs its own index because Postgres treats
|
|
72
|
+
-- NULLs as distinct in a unique index, so the one above would happily allow a second
|
|
73
|
+
-- open row for the same box.
|
|
74
|
+
CREATE UNIQUE INDEX IF NOT EXISTS "uq_service_incident_open_server"
|
|
75
|
+
ON "service_incident" ("server_id")
|
|
76
|
+
WHERE "status" = 'open' AND "project_id" IS NULL;
|
|
77
|
+
--> statement-breakpoint
|
|
78
|
+
CREATE INDEX IF NOT EXISTS "idx_service_incident_project"
|
|
79
|
+
ON "service_incident" ("project_id", "opened_at" DESC);
|
|
80
|
+
--> statement-breakpoint
|
|
81
|
+
-- Intent marker. `disableProject` stopped the container and recorded NOTHING, so
|
|
82
|
+
-- "the operator turned this off" was unknowable — the health watch would page about
|
|
83
|
+
-- every deliberately-disabled project, forever. Set on disable, cleared on enable.
|
|
84
|
+
ALTER TABLE "project" ADD COLUMN IF NOT EXISTS "disabled_at" timestamp;
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
-- Widen the daily analytics rollup: distinct visitors, top paths, status mix.
|
|
2
|
+
--
|
|
3
|
+
-- The visitor pipeline already persisted per-country counts, and the dashboard
|
|
4
|
+
-- already rendered cards for "unique IPs" and a top-paths table. Neither was real:
|
|
5
|
+
-- `top_paths` was hardcoded `[]` in the service layer, and "unique IPs" was
|
|
6
|
+
-- `unique_requests` — the count of non-static REQUESTS — so five page views from
|
|
7
|
+
-- one browser read as five visitors. These columns are what makes those two
|
|
8
|
+
-- numbers mean what the UI has been claiming.
|
|
9
|
+
--
|
|
10
|
+
-- All three land HERE, on the daily table, rather than on the per-minute one:
|
|
11
|
+
-- visitors and paths are the high-cardinality metrics, and holding them per-minute
|
|
12
|
+
-- would multiply the edge's shared-dict key count by ~1440 to feed a chart nobody
|
|
13
|
+
-- asked for — evicting, in the process, the request counters they annotate.
|
|
14
|
+
-- Countries already live here for the same reason, and the edge keeps all four
|
|
15
|
+
-- under one `g:{domain}:{day}:` key prefix so a reader gets the whole day in a
|
|
16
|
+
-- single dict scan.
|
|
17
|
+
--
|
|
18
|
+
-- Additive with defaults on purpose: `migrations-additive.test.ts` rejects a
|
|
19
|
+
-- NOT NULL column without a DEFAULT, because an older cross-version dump omits the
|
|
20
|
+
-- column entirely and Drizzle then emits DEFAULT for it — with no default that's a
|
|
21
|
+
-- NULL into NOT NULL on the newer receiver, i.e. a broken cloud/project transfer.
|
|
22
|
+
|
|
23
|
+
-- Distinct visitors for the day, deduplicated AT THE EDGE and stored as a count.
|
|
24
|
+
--
|
|
25
|
+
-- Privacy is a property of where the dedup happens: the edge hashes each address
|
|
26
|
+
-- with a salt that is generated on that box, rotates daily, and never leaves its
|
|
27
|
+
-- shared memory. Only the cardinality of that set is written here. No address and
|
|
28
|
+
-- no per-visitor row exists at any layer, which is what keeps the published
|
|
29
|
+
-- "we don't collect behavioural analytics on your end users" true while still
|
|
30
|
+
-- reporting a real visitor number.
|
|
31
|
+
--
|
|
32
|
+
-- Understates on a very busy domain: the edge's `visitors` zone LRU-evicts past
|
|
33
|
+
-- roughly 1M distinct/day. mgmt_api `GET /status` reports that zone's free space so
|
|
34
|
+
-- a reader can label the number approximate instead of presenting an eviction
|
|
35
|
+
-- artifact as a measurement.
|
|
36
|
+
ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "visitors" integer DEFAULT 0 NOT NULL;
|
|
37
|
+
--> statement-breakpoint
|
|
38
|
+
|
|
39
|
+
-- Top paths: { "/": 900, "/orders/:id": 120, "other": 40 }.
|
|
40
|
+
--
|
|
41
|
+
-- Normalized at the edge before it ever becomes a key: query string stripped (so a
|
|
42
|
+
-- ?token= or ?session= can never be persisted here), numeric and UUID segments
|
|
43
|
+
-- collapsed to `:id`, key length capped, and the tail past a cardinality cap folded
|
|
44
|
+
-- into "other" — otherwise a scanner walking /wp-admin variants mints thousands of
|
|
45
|
+
-- permanent keys and evicts the real counters.
|
|
46
|
+
ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "paths" jsonb;
|
|
47
|
+
--> statement-breakpoint
|
|
48
|
+
|
|
49
|
+
-- Status-code mix: { "200": 4210, "404": 17, "502": 3 }. Status was captured into
|
|
50
|
+
-- the edge's raw-request ring buffer (RAM, 1h) but never aggregated, so error rate
|
|
51
|
+
-- was unanswerable from any persisted data. Bounded by construction.
|
|
52
|
+
ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "statuses" jsonb;
|
|
53
|
+
--> statement-breakpoint
|
|
54
|
+
|
|
55
|
+
-- Widen the per-minute bandwidth counters from int4 to int8.
|
|
56
|
+
--
|
|
57
|
+
-- These are bytes for ONE MINUTE, and int4 caps at 2,147,483,647 — about 2.1 GB,
|
|
58
|
+
-- which a single minute reaches at roughly 286 Mbps. That is unremarkable for a
|
|
59
|
+
-- site serving video or large downloads. Past it Postgres raises "integer out of
|
|
60
|
+
-- range", the upsert for that minute fails, and the scrape for the whole domain
|
|
61
|
+
-- dies with it — so the busiest domains on a box would be precisely the ones with
|
|
62
|
+
-- no analytics, and it would look like the feature simply didn't work for them.
|
|
63
|
+
--
|
|
64
|
+
-- Now that collection runs unattended on a schedule rather than only when someone
|
|
65
|
+
-- opens the tab, that failure would happen in a job with nobody reading the error.
|
|
66
|
+
--
|
|
67
|
+
-- Widening is lossless and there is no application change: the edge already
|
|
68
|
+
-- accumulates these as Lua numbers (doubles, exact to 2^53) and Drizzle reads them
|
|
69
|
+
-- back with mode:"number", which is exact over the same range.
|
|
70
|
+
ALTER TABLE "server_analytics" ALTER COLUMN "bandwidth_in" TYPE bigint;
|
|
71
|
+
--> statement-breakpoint
|
|
72
|
+
ALTER TABLE "server_analytics" ALTER COLUMN "bandwidth_out" TYPE bigint;
|