openship 0.5.0 → 0.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -1
- package/dist/bare-IWS2QHJ3.js +23 -0
- package/dist/{bowser-O3AEA6AN.js → bowser-HNOTPGJO.js} +2 -2
- package/dist/{chunk-5MZ624VX.js → chunk-2GWYYHOR.js} +1 -1
- package/dist/{chunk-BS2AXYZX.js → chunk-2NZSUXOB.js} +647 -214
- package/dist/chunk-3FYIXVX5.js +8051 -0
- package/dist/{chunk-ILFETZOM.js → chunk-3LDWWY7U.js} +21 -1
- package/dist/{chunk-W73X43R7.js → chunk-3NQQXMDR.js} +3 -3
- package/dist/{chunk-NNPEG6HM.js → chunk-3NZICQHF.js} +2 -2
- package/dist/{chunk-OWED2GRM.js → chunk-42WQABXM.js} +2 -2
- package/dist/{chunk-TYVME3XD.js → chunk-6FH4AXVY.js} +20 -3
- package/dist/{chunk-ZCOQJCZO.js → chunk-6ODUBKUB.js} +3 -3
- package/dist/{chunk-DYPH4HCI.js → chunk-AO252K7Z.js} +1 -1
- package/dist/{chunk-IXCHURVK.js → chunk-DOD2BXHQ.js} +4 -4
- package/dist/{chunk-VWPEKOH6.js → chunk-DRK3MEDS.js} +3 -3
- package/dist/{chunk-V7BZMGOT.js → chunk-EMXI4UCU.js} +5 -5
- package/dist/{chunk-RB35YN3J.js → chunk-EW6KB2J7.js} +391 -63
- package/dist/chunk-I3DPMTSF.js +29 -0
- package/dist/{chunk-NXFIRK2Z.js → chunk-IOT5ZX23.js} +573 -459
- package/dist/{chunk-KZOBYJUW.js → chunk-IW2UKBVF.js} +3 -3
- package/dist/{chunk-YWMODKWL.js → chunk-IXM5245F.js} +165 -13
- package/dist/{chunk-KNTJ7PSF.js → chunk-JIMUYEKN.js} +4 -4
- package/dist/{chunk-O6ZZF52V.js → chunk-ONKEYSMY.js} +2 -2
- package/dist/{chunk-UMQP7SYT.js → chunk-OTLSJYAC.js} +1316 -53
- package/dist/{chunk-EI575GHN.js → chunk-PDW4L3NV.js} +3 -3
- package/dist/{chunk-KI2EY3WB.js → chunk-PXKCZB2I.js} +1 -1
- package/dist/{chunk-6APIPIJY.js → chunk-QWHBRY2K.js} +5 -5
- package/dist/{chunk-JLKKYUZA.js → chunk-RRJJGVKE.js} +7 -7
- package/dist/{chunk-6B5YCFFV.js → chunk-RVER2EOO.js} +5 -5
- package/dist/{chunk-Q2CIRAOK.js → chunk-RXQBD4ZY.js} +1 -1
- package/dist/{chunk-6LKHXB6T.js → chunk-TWJCLD4W.js} +48 -13
- package/dist/{chunk-K2RPVGPN.js → chunk-UC7YIEHT.js} +36 -17
- package/dist/{chunk-I6PSTQIR.js → chunk-VDDMADOF.js} +3 -3
- package/dist/{chunk-6PIYRG5I.js → chunk-VZ2ZHPS2.js} +52 -42
- package/dist/{chunk-AVIB7CT7.js → chunk-WNWQBB7F.js} +2 -2
- package/dist/{chunk-FRNAR33O.js → chunk-XZYFZ6A5.js} +3 -3
- package/dist/chunk-YIPMDGG6.js +703 -0
- package/dist/{chunk-RPHRPFEH.js → chunk-ZCFVHA25.js} +1 -1
- package/dist/{chunk-FPRHYBY2.js → chunk-ZFJXPCQP.js} +1 -1
- package/dist/{chunk-JYGD5WSN.js → chunk-ZWLDLTJP.js} +1 -1
- package/dist/{cloud-7QYOGF7F.js → cloud-KRTZ2IL5.js} +3 -3
- package/dist/cloud-M2YLWEXO.js +31 -0
- package/dist/{detect-BAJMY6KM.js → detect-ALF4GXRX.js} +7 -8
- package/dist/{dist-IQJ6H6UZ.js → dist-BM6NPWIH.js} +3 -3
- package/dist/{dist-es-X3B5QYKB.js → dist-es-2YRDXMUU.js} +6 -6
- package/dist/{dist-es-OSNJBLXB.js → dist-es-CCVZEFRZ.js} +5 -5
- package/dist/{dist-es-VNKHQVIB.js → dist-es-IOH6G4GM.js} +4 -4
- package/dist/{dist-es-RYHIPOTN.js → dist-es-JGBPABJI.js} +14 -14
- package/dist/{dist-es-JSTDUS7Q.js → dist-es-KWSMDPIP.js} +6 -6
- package/dist/{dist-es-5DY7K32L.js → dist-es-UYDDOVZ3.js} +6 -6
- package/dist/{dist-es-WV4AMOII.js → dist-es-XUOR4NQN.js} +8 -8
- package/dist/docker-QECQJ5P4.js +32 -0
- package/dist/docker-edge-executor-62N64MGS.js +17 -0
- package/dist/edge-import-SAYTRJLO.js +48 -0
- package/dist/ensure-container-edge-CFSCPGZK.js +32 -0
- package/dist/{event-streams-EHRAKL75.js → event-streams-TR4GAC53.js} +4 -4
- package/dist/{executor-LOYMPU67.js → executor-PT6UEJ5M.js} +8 -8
- package/dist/index.js +1228 -463
- package/dist/{loadSso-46FQRIZS.js → loadSso-UGF4JMRK.js} +10 -10
- package/dist/{local-executor-V7U5VOZW.js → local-executor-WOFZD72X.js} +4 -4
- package/dist/nginx-YOEEFBTE.js +31 -0
- package/dist/node-entry.js +4 -0
- package/dist/{noop-FMMFG2EZ.js → noop-NJ72F46T.js} +3 -3
- package/dist/openresty-lua-WGBO2M3Z.js +81 -0
- package/dist/server/index.js +23276 -8015
- package/dist/server/lua/geo_country.lua +34 -8
- package/dist/server/lua/maxminddb.lua +403 -0
- package/dist/server/lua/mgmt_api.lua +431 -68
- package/dist/server/lua/pipe_stream.lua +52 -39
- package/dist/server/lua/site_logger.lua +236 -30
- package/dist/server/migrations/0082_edge_target_verification.sql +39 -0
- package/dist/server/migrations/0083_service_incident.sql +84 -0
- package/dist/server/migrations/0084_analytics_daily_rollup.sql +72 -0
- package/dist/server/migrations/0085_resource_usage.sql +68 -0
- package/dist/server/migrations/0086_remote_infra_updates.sql +41 -0
- package/dist/server/migrations/0087_server_container_version_columns.sql +17 -0
- package/dist/server/migrations/0088_audit_source_and_settings.sql +27 -0
- package/dist/server/migrations/0089_auto_scan_infra.sql +7 -0
- package/dist/server/migrations/0090_project_collect_paths.sql +18 -0
- package/dist/server/migrations/0091_project_server_id.sql +26 -0
- package/dist/server/migrations/0092_project_internal_alias.sql +13 -0
- package/dist/server/migrations/0093_service_incident_org_index.sql +13 -0
- package/dist/server/migrations/0094_backup_restore_meta.sql +15 -0
- package/dist/server/migrations/0095_backup_restore_cancel.sql +19 -0
- package/dist/server/migrations/0096_backup_policy_retention_defaults.sql +20 -0
- package/dist/server/migrations/0097_update_status_upstream_only.sql +35 -0
- package/dist/server/migrations/meta/_journal.json +112 -0
- package/dist/setup-M2ZDGH3O.js +21 -0
- package/dist/{signin-ITOPKKUB.js → signin-3M3UPB6K.js} +10 -10
- package/dist/{sso-oidc-NON2PD7Y.js → sso-oidc-3VDIYOEJ.js} +10 -10
- package/dist/{sts-HHFFSAY2.js → sts-B3WIYLXR.js} +9 -9
- package/package.json +6 -3
- package/dist/bare-VRGE5MXH.js +0 -23
- package/dist/chunk-CUJXGK7R.js +0 -3167
- package/dist/chunk-EAAGSQXN.js +0 -377
- package/dist/cloud-3M2WZ624.js +0 -30
- package/dist/docker-2NGBRJFJ.js +0 -18
- package/dist/docker-edge-executor-5I4M5QT7.js +0 -13
- package/dist/edge-import-RIWNEIXQ.js +0 -47
- package/dist/ensure-container-edge-3NSYKHRX.js +0 -31
- package/dist/nginx-QE4DOM5J.js +0 -24
- package/dist/openresty-lua-BTAKECCF.js +0 -42
- package/dist/server/lua/pipe_log.lua +0 -71
- package/dist/setup-KELALK7H.js +0 -20
|
@@ -2,6 +2,14 @@
|
|
|
2
2
|
-- content_by_lua: SSE endpoint for real-time request log streaming.
|
|
3
3
|
-- GET /logs/stream?domain=example.com
|
|
4
4
|
-- Internal only - 127.0.0.1:9145
|
|
5
|
+
--
|
|
6
|
+
-- Reads the SAME ring buffer site_logger.lua writes (rlog:{domain}:seq +
|
|
7
|
+
-- rlog:{domain}:{slot}), by a PER-CONNECTION cursor. Nothing is consumed: N
|
|
8
|
+
-- concurrent watchers each walk the ring independently and none steals another's
|
|
9
|
+
-- frames. This is what replaced the old single-subscriber log_pipe:q queue —
|
|
10
|
+
-- a containerized edge tails this over `docker exec curl`, and a curl that
|
|
11
|
+
-- outlives its client (the daemon buffers its stdout, so no EPIPE) can no longer
|
|
12
|
+
-- drain the queue out from under the next reader.
|
|
5
13
|
|
|
6
14
|
local sh = ngx.shared.request_data
|
|
7
15
|
if not sh then
|
|
@@ -20,12 +28,10 @@ end
|
|
|
20
28
|
domain = domain:lower()
|
|
21
29
|
if domain:sub(1, 4) == "www." then domain = domain:sub(5) end
|
|
22
30
|
|
|
23
|
-
|
|
24
|
-
local
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
sh:delete(QUEUE_KEY)
|
|
28
|
-
sh:set(SUB_KEY, true, 30)
|
|
31
|
+
-- MUST match site_logger.lua's RING — same slot arithmetic on both ends.
|
|
32
|
+
local RING = 1000
|
|
33
|
+
local SEQ_KEY = "rlog:" .. domain .. ":seq"
|
|
34
|
+
local slot_key = function(seq) return "rlog:" .. domain .. ":" .. (seq % RING) end
|
|
29
35
|
|
|
30
36
|
ngx.header["Content-Type"] = "text/event-stream"
|
|
31
37
|
ngx.header["Cache-Control"] = "no-cache, no-store"
|
|
@@ -36,55 +42,62 @@ ngx.header["X-Accel-Buffering"] = "no"
|
|
|
36
42
|
-- Without this, ngx.flush on an empty buffer sends chunked-EOF (0\r\n\r\n)
|
|
37
43
|
-- and terminates the response immediately.
|
|
38
44
|
ngx.print(": connected\n\n")
|
|
39
|
-
if not ngx.flush(true) then
|
|
40
|
-
sh:delete(SUB_KEY)
|
|
41
|
-
return
|
|
42
|
-
end
|
|
45
|
+
if not ngx.flush(true) then return end
|
|
43
46
|
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
+
-- Start at the current head: stream only requests that arrive AFTER connect.
|
|
48
|
+
-- History up to this point is served separately by /logs/recent, and the client
|
|
49
|
+
-- dedups the two by the deterministic `{host}:{seq}` id, so a row seen by both
|
|
50
|
+
-- collapses to one.
|
|
51
|
+
local cursor = tonumber(sh:get(SEQ_KEY)) or 0
|
|
52
|
+
local started = ngx.now()
|
|
53
|
+
local last_hb = started
|
|
47
54
|
|
|
48
55
|
while true do
|
|
49
|
-
-- Max 1 hour per connection
|
|
50
|
-
if ngx.now() - started > 3600 then
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
56
|
+
-- Max 1 hour per connection.
|
|
57
|
+
if ngx.now() - started > 3600 then return end
|
|
58
|
+
|
|
59
|
+
local head = tonumber(sh:get(SEQ_KEY)) or 0
|
|
60
|
+
|
|
61
|
+
-- If we've fallen more than a full ring behind, the oldest unread slots have
|
|
62
|
+
-- already been overwritten. Jump to the oldest slot still holding its own
|
|
63
|
+
-- entry (head-RING maps to head's slot, so head-RING+1 is the oldest live one)
|
|
64
|
+
-- and skip the lost range — same bound the old 2000-entry queue had.
|
|
65
|
+
if cursor < head - RING then cursor = head - RING end
|
|
54
66
|
|
|
55
|
-
-- Drain up to 100 queued entries per cycle
|
|
56
67
|
local sent = 0
|
|
57
|
-
while sent < 100 do
|
|
58
|
-
local
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
68
|
+
while cursor < head and sent < 100 do
|
|
69
|
+
local next_seq = cursor + 1
|
|
70
|
+
local entry = sh:get(slot_key(next_seq))
|
|
71
|
+
if entry then
|
|
72
|
+
ngx.print("event: request\ndata: ", entry, "\n\n")
|
|
73
|
+
sent = sent + 1
|
|
74
|
+
cursor = next_seq
|
|
75
|
+
elseif next_seq < head then
|
|
76
|
+
-- A later seq already exists, so this can't be the incr(seq)→set(slot)
|
|
77
|
+
-- write window: the slot is genuinely lost (expired or wrapped). Skip it.
|
|
78
|
+
cursor = next_seq
|
|
79
|
+
else
|
|
80
|
+
-- next_seq == head: the newest entry's slot may not be written yet
|
|
81
|
+
-- (the microsecond window between incr(seq) and set(slot)). Retry it
|
|
82
|
+
-- next cycle without advancing.
|
|
83
|
+
break
|
|
84
|
+
end
|
|
62
85
|
end
|
|
63
86
|
|
|
64
87
|
if sent > 0 then
|
|
65
|
-
if not ngx.flush(true) then
|
|
66
|
-
sh:delete(SUB_KEY)
|
|
67
|
-
return
|
|
68
|
-
end
|
|
88
|
+
if not ngx.flush(true) then return end
|
|
69
89
|
end
|
|
70
90
|
|
|
71
91
|
local now = ngx.now()
|
|
72
92
|
|
|
73
|
-
-- Heartbeat every 15s
|
|
93
|
+
-- Heartbeat every 15s. Its flush is also how a dead client is detected: once
|
|
94
|
+
-- the reader (a `docker exec curl`) goes away and its socket closes, this
|
|
95
|
+
-- flush fails and the connection tears down here.
|
|
74
96
|
if now - last_hb > 15 then
|
|
75
97
|
ngx.print(": ping\n\n")
|
|
76
|
-
if not ngx.flush(true) then
|
|
77
|
-
sh:delete(SUB_KEY)
|
|
78
|
-
return
|
|
79
|
-
end
|
|
98
|
+
if not ngx.flush(true) then return end
|
|
80
99
|
last_hb = now
|
|
81
100
|
end
|
|
82
101
|
|
|
83
|
-
-- Refresh subscriber TTL every 10s
|
|
84
|
-
if now - last_ref > 10 then
|
|
85
|
-
sh:set(SUB_KEY, true, 30)
|
|
86
|
-
last_ref = now
|
|
87
|
-
end
|
|
88
|
-
|
|
89
102
|
ngx.sleep(0.05)
|
|
90
103
|
end
|
|
@@ -8,17 +8,33 @@
|
|
|
8
8
|
-- s:{domain}:{epoch_min}:i bandwidth in bytes (TTL 24h)
|
|
9
9
|
-- s:{domain}:{epoch_min}:o bandwidth out bytes (TTL 24h)
|
|
10
10
|
-- s:{domain}:{epoch_min}:t response time sum (seconds) (TTL 24h)
|
|
11
|
-
-- s:{domain}:{epoch_min}:u
|
|
12
|
-
-- g:{domain}:{YYYYMMDD}:{CC} country hit count (TTL 48h)
|
|
11
|
+
-- s:{domain}:{epoch_min}:u page (non-static) requests (TTL 24h)
|
|
13
12
|
-- c:{domain}:{epoch_min}:{CC} country per minute (TTL 24h)
|
|
14
13
|
-- t:{domain}:r / :i / :o lifetime totals (no TTL)
|
|
15
14
|
-- d:{domain} domain index marker (no TTL)
|
|
16
15
|
--
|
|
16
|
+
-- Daily rollup, all under one `g:{domain}:{YYYYMMDD}:` prefix so mgmt_api's
|
|
17
|
+
-- /analytics/geo serves the whole day in a SINGLE dict scan:
|
|
18
|
+
-- g:{domain}:{YYYYMMDD}:{CC} country hit count (TTL 48h)
|
|
19
|
+
-- g:{domain}:{YYYYMMDD}:v distinct visitors (TTL 48h)
|
|
20
|
+
-- g:{domain}:{YYYYMMDD}:p:{path} path hit count (TTL 48h)
|
|
21
|
+
-- g:{domain}:{YYYYMMDD}:s:{code} status-code count (TTL 48h)
|
|
22
|
+
--
|
|
23
|
+
-- Shared dict key schema (visitors zone):
|
|
24
|
+
-- vsalt:{YYYYMMDD} per-day hash salt (TTL 48h)
|
|
25
|
+
-- vd:{domain}:{YYYYMMDD}:{hash} distinct-visitor marker (TTL 48h)
|
|
26
|
+
--
|
|
17
27
|
-- Shared dict key schema (request_data zone):
|
|
18
28
|
-- rlog:{domain}:seq monotonic write pointer
|
|
19
29
|
-- rlog:{domain}:{slot} JSON entry (ring buf) (TTL 1h)
|
|
20
|
-
--
|
|
21
|
-
--
|
|
30
|
+
--
|
|
31
|
+
-- The ring is ALSO the live source: pipe_stream.lua's SSE endpoint reads these
|
|
32
|
+
-- slots by a per-connection cursor, so there is no separate live queue or
|
|
33
|
+
-- subscriber flag — every watcher sees the full stream and steals from none.
|
|
34
|
+
--
|
|
35
|
+
-- NOTE on `:u` — it counts non-static REQUESTS, not people. It was surfaced as
|
|
36
|
+
-- "unique IPs" all the way to the dashboard, which it never was. Real distinct
|
|
37
|
+
-- visitors are the `vd:`/`:v` pair below.
|
|
22
38
|
|
|
23
39
|
local cjson = require "cjson.safe"
|
|
24
40
|
|
|
@@ -29,12 +45,17 @@ if not geo_ok then
|
|
|
29
45
|
ngx.log(ngx.WARN, "[site_logger] geo_country module not available. GeoIP disabled.")
|
|
30
46
|
end
|
|
31
47
|
|
|
32
|
-
-- SAFE LOAD: Pipe module (per-worker, cached by require)
|
|
33
|
-
local pipe_ok, pipe_log = pcall(require, "openship.pipe_log")
|
|
34
|
-
if not pipe_ok then pipe_log = nil end
|
|
35
|
-
|
|
36
48
|
local analytics = ngx.shared.analytics
|
|
37
49
|
local request_data = ngx.shared.request_data
|
|
50
|
+
-- Per-host edge config, pushed by the API (see mgmt_api's /analytics/config).
|
|
51
|
+
--
|
|
52
|
+
-- The `rules` zone, NOT `analytics`: this holds a handful of small config values and is
|
|
53
|
+
-- never under eviction pressure, whereas the analytics zone is a 256 MB churn of counters.
|
|
54
|
+
-- An evicted flag there would silently stop path collection with nothing to explain why.
|
|
55
|
+
local rules_dict = ngx.shared.rules
|
|
56
|
+
-- Optional: a box whose nginx.conf predates this zone keeps working, just without
|
|
57
|
+
-- distinct-visitor counts. Never assume a dict exists.
|
|
58
|
+
local visitors = ngx.shared.visitors
|
|
38
59
|
if not analytics or not request_data then return end
|
|
39
60
|
|
|
40
61
|
-- ── Helpers ──────────────────────────────────────────────────────────────────
|
|
@@ -89,11 +110,71 @@ local function is_static_asset(u)
|
|
|
89
110
|
return false
|
|
90
111
|
end
|
|
91
112
|
|
|
113
|
+
-- Collapse a request URI into a low-cardinality bucket suitable as a dict key.
|
|
114
|
+
-- Without this, "top paths" is unbounded: every ?token=, every /orders/48219, and
|
|
115
|
+
-- every 404 a scanner probes mints a permanent key, and the analytics zone
|
|
116
|
+
-- LRU-evicts real counters to store junk.
|
|
117
|
+
local function normalize_path(u)
|
|
118
|
+
-- Query string carries ids/tokens and never identifies the route.
|
|
119
|
+
local path = u:match("^([^?#]*)") or u
|
|
120
|
+
if path == "" then return "/" end
|
|
121
|
+
-- Numeric and UUID/hex segments are ids, not routes: /orders/48219 and
|
|
122
|
+
-- /orders/48220 are one path.
|
|
123
|
+
path = path:gsub("/%d+", "/:id")
|
|
124
|
+
path = path:gsub("/%x%x%x%x%x%x%x%x%-?%x*%-?%x*%-?%x*%-?%x*", "/:id")
|
|
125
|
+
-- Bound the key itself — a long URI must not become a long key.
|
|
126
|
+
if #path > 120 then path = path:sub(1, 120) .. "…" end
|
|
127
|
+
return path
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
-- ── Enumerable sets, without a zone scan ─────────────────────────────────────
|
|
131
|
+
--
|
|
132
|
+
-- Reading these counters back used to need key DISCOVERY, and a shared dict offers exactly
|
|
133
|
+
-- one mechanism for that: `get_keys`, which walks EVERY key in the zone holding its lock.
|
|
134
|
+
-- Cost scaled with how much was in the 256 MB zone, not with what was being asked for, and
|
|
135
|
+
-- it ran on the mgmt path four different ways.
|
|
136
|
+
--
|
|
137
|
+
-- So every unbounded set now carries its own index: a counter plus numbered slots.
|
|
138
|
+
--
|
|
139
|
+
-- <counter> how many distinct members
|
|
140
|
+
-- <prefix><n> the n-th member
|
|
141
|
+
--
|
|
142
|
+
-- A reader reads the counter and then reads slots 1..n. Bounded by the answer.
|
|
143
|
+
--
|
|
144
|
+
-- The write only happens on a member's FIRST sighting, so steady state costs nothing: the
|
|
145
|
+
-- millionth request from Germany today takes the plain `incr` path and touches no index.
|
|
146
|
+
local function index_add(counter_key, slot_prefix, member, ttl)
|
|
147
|
+
local n = analytics:incr(counter_key, 1, 0, ttl)
|
|
148
|
+
if n then analytics:set(slot_prefix .. n, member, ttl) end
|
|
149
|
+
end
|
|
150
|
+
|
|
151
|
+
--- Add 1 to a counter, creating it (with TTL) and indexing it on first sighting.
|
|
152
|
+
--
|
|
153
|
+
-- `incr` with no `init` returns nil for an absent key, which doubles as the existence
|
|
154
|
+
-- probe — so the common case is ONE hash operation. TTL is set by the creating `safe_add`
|
|
155
|
+
-- rather than passed to `incr`: shdict rejects `init_ttl` without `init`, and that raises a
|
|
156
|
+
-- Lua error which aborts the whole log handler (it has silently zeroed every counter below
|
|
157
|
+
-- it before).
|
|
158
|
+
local function bump_indexed(key, ttl, counter_key, slot_prefix, member)
|
|
159
|
+
if analytics:incr(key, 1) then return end
|
|
160
|
+
if analytics:safe_add(key, 1, ttl) then
|
|
161
|
+
index_add(counter_key, slot_prefix, member, ttl)
|
|
162
|
+
else
|
|
163
|
+
-- Another worker created it between our probe and our add; our +1 still counts.
|
|
164
|
+
analytics:incr(key, 1)
|
|
165
|
+
end
|
|
166
|
+
end
|
|
167
|
+
|
|
92
168
|
local RING = 1000
|
|
93
169
|
local D24H = 86400
|
|
94
170
|
local D48H = 172800
|
|
95
171
|
local D1H = 3600
|
|
96
172
|
|
|
173
|
+
-- Distinct paths tracked per domain per day. Past this the tail folds into
|
|
174
|
+
-- "other" rather than growing the zone — a URL-fuzzing bot is capped at one
|
|
175
|
+
-- extra key, not thousands.
|
|
176
|
+
local PATH_CARDINALITY_CAP = 2000
|
|
177
|
+
|
|
97
178
|
-- ── Capture ──────────────────────────────────────────────────────────────────
|
|
98
179
|
|
|
99
180
|
local host = normalize(ngx.var.host)
|
|
@@ -109,6 +190,23 @@ local rt = tonumber(ngx.var.request_time) or 0 -- seconds (float)
|
|
|
109
190
|
local method = ngx.var.request_method or "GET"
|
|
110
191
|
local status = ngx.var.status or "0"
|
|
111
192
|
|
|
193
|
+
-- Client country, resolved ONCE per request.
|
|
194
|
+
--
|
|
195
|
+
-- Hoisted out of the geo block because both the daily/per-minute rollups and the raw
|
|
196
|
+
-- ring buffer need it. The ring is the single source for the request log — /logs/recent
|
|
197
|
+
-- reads it directly and the live SSE stream (pipe_stream.lua) reads the same slots by
|
|
198
|
+
-- cursor — so resolving the country here means every row carries its flag no matter
|
|
199
|
+
-- which path delivered it. An earlier split (looked up in the rollup, again in a
|
|
200
|
+
-- separate live pipe) left the ring with no country at all.
|
|
201
|
+
--
|
|
202
|
+
-- pcall'd because a corrupt or partially-written mmdb makes the lookup raise, and an
|
|
203
|
+
-- error here would abort every counter below it (see the section header).
|
|
204
|
+
local cc = nil
|
|
205
|
+
if geo and geo.get_country_code then
|
|
206
|
+
local ok_cc, res = pcall(geo.get_country_code, ip)
|
|
207
|
+
if ok_cc and type(res) == "string" and res ~= "" then cc = res end
|
|
208
|
+
end
|
|
209
|
+
|
|
112
210
|
-- ── 1. Minute-bucket counters ────────────────────────────────────────────────
|
|
113
211
|
|
|
114
212
|
local minute = math.floor(ts / 60)
|
|
@@ -136,26 +234,145 @@ analytics:incr("t:" .. host .. ":r", 1, 0)
|
|
|
136
234
|
analytics:incr("t:" .. host .. ":i", req_len, 0)
|
|
137
235
|
analytics:incr("t:" .. host .. ":o", bytes, 0)
|
|
138
236
|
|
|
139
|
-
-- Domain index
|
|
140
|
-
|
|
237
|
+
-- Domain index. `safe_add` returning true means this is the first request this box has
|
|
238
|
+
-- ever seen for the host, which is exactly when the enumerable index needs appending —
|
|
239
|
+
-- so domain discovery no longer costs a zone scan either.
|
|
240
|
+
if analytics:safe_add("d:" .. host, 1) then
|
|
241
|
+
index_add("dn", "di:", host)
|
|
242
|
+
end
|
|
141
243
|
|
|
142
|
-
-- ── 3.
|
|
244
|
+
-- ── 3. Daily rollup: geo, visitors, paths, statuses ──────────────────────────
|
|
245
|
+
--
|
|
246
|
+
-- Each sub-block is pcall'd SEPARATELY, and that is load-bearing rather than
|
|
247
|
+
-- defensive habit. A raised error in log_by_lua aborts the remainder of the
|
|
248
|
+
-- handler, so one bad shdict call silently zeroes every counter written after it —
|
|
249
|
+
-- which is exactly what an `incr(..., nil, ttl)` here did: paths AND statuses read
|
|
250
|
+
-- as "no traffic" while requests and bandwidth looked perfect. Isolating them
|
|
251
|
+
-- means a future edit can lose at most its own metric.
|
|
252
|
+
|
|
253
|
+
local day = today()
|
|
254
|
+
local gpfx = "g:" .. host .. ":" .. day .. ":"
|
|
143
255
|
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
256
|
+
-- 3a. GeoIP. `cc` was resolved once above and is reused by the ring buffer and the
|
|
257
|
+
-- live pipe, so a request costs exactly one mmdb lookup no matter how many consumers
|
|
258
|
+
-- report its country.
|
|
259
|
+
if cc then
|
|
260
|
+
pcall(function()
|
|
147
261
|
-- Daily geo (for /analytics/geo endpoint)
|
|
148
|
-
|
|
149
|
-
-- Per-minute geo (for time-series country breakdown)
|
|
150
|
-
|
|
262
|
+
bump_indexed(gpfx .. cc, D48H, gpfx .. "cn", gpfx .. "ci:", cc)
|
|
263
|
+
-- Per-minute geo (for the time-series country breakdown). Indexed per MINUTE, so
|
|
264
|
+
-- the scraper reads only the minutes in its window.
|
|
265
|
+
local mpfx = "c:" .. host .. ":" .. minute .. ":"
|
|
266
|
+
bump_indexed(mpfx .. cc, D24H, mpfx .. "n", mpfx .. "i:", cc)
|
|
267
|
+
end)
|
|
268
|
+
end
|
|
269
|
+
|
|
270
|
+
-- 3b. Distinct visitors — a COUNT, never an identity.
|
|
271
|
+
--
|
|
272
|
+
-- The address is hashed with a salt that is generated on this box, rotates daily,
|
|
273
|
+
-- and is never written anywhere but this shared dict, so the marker keys are not
|
|
274
|
+
-- reversible to an IP and are worthless the next day. Only the resulting counter
|
|
275
|
+
-- is ever flushed to Postgres; no per-visitor row exists at any layer. That is
|
|
276
|
+
-- what lets us report real visitor numbers while still not collecting behavioural
|
|
277
|
+
-- analytics on anyone's end users.
|
|
278
|
+
--
|
|
279
|
+
-- Per DAY, not per minute: a per-minute set is ~1440x the keys for a number
|
|
280
|
+
-- nobody reads, and would evict the counters it exists to annotate.
|
|
281
|
+
pcall(function()
|
|
282
|
+
if visitors then
|
|
283
|
+
local salt_key = "vsalt:" .. day
|
|
284
|
+
local salt = visitors:get(salt_key)
|
|
285
|
+
if not salt then
|
|
286
|
+
-- First writer wins; a racing worker's safe_add fails and it re-reads. The
|
|
287
|
+
-- salt only has to be unpredictable and stable for the day.
|
|
288
|
+
salt = tostring(ngx.now()) .. tostring(math.random(1, 2147483647))
|
|
289
|
+
local ok = visitors:safe_add(salt_key, salt, D48H)
|
|
290
|
+
if not ok then salt = visitors:get(salt_key) or salt end
|
|
291
|
+
end
|
|
292
|
+
-- safe_add is the dedup: it succeeds ONLY on this visitor's first request
|
|
293
|
+
-- today, so the counter below increments exactly once per distinct visitor.
|
|
294
|
+
local marker = "vd:" .. host .. ":" .. day .. ":" .. ngx.crc32_long(salt .. ip)
|
|
295
|
+
if visitors:safe_add(marker, 1, D48H) then
|
|
296
|
+
analytics:incr(gpfx .. "v", 1, 0, D48H)
|
|
297
|
+
end
|
|
298
|
+
end
|
|
299
|
+
end)
|
|
300
|
+
|
|
301
|
+
-- 3c. Top paths — OPT-IN, non-static only, normalized and cardinality-capped.
|
|
302
|
+
--
|
|
303
|
+
-- Off unless the project turned it on, because this one block is the most expensive thing
|
|
304
|
+
-- the log handler does. Measured on the shipped edge image, per request:
|
|
305
|
+
--
|
|
306
|
+
-- paths (whole block) 1.72 us <- 57% of the counter path
|
|
307
|
+
-- of which string work 1.38 us normalize_path + is_static_asset
|
|
308
|
+
-- minute buckets (5 incr) 0.61 us
|
|
309
|
+
-- country (2 incr) 0.24 us
|
|
310
|
+
-- status (1 incr) 0.14 us
|
|
311
|
+
-- this flag read (1 get) 0.07 us
|
|
312
|
+
--
|
|
313
|
+
-- So the gate costs 0.07 us and saves 1.72 us — the counter path drops from ~3.0 us to
|
|
314
|
+
-- ~1.35 us with paths off. It is also the highest-cardinality dimension (up to
|
|
315
|
+
-- PATH_CARDINALITY_CAP keys per domain per day, against ~200 countries and a few dozen
|
|
316
|
+
-- statuses) and the largest column in the daily rollup.
|
|
317
|
+
--
|
|
318
|
+
-- Absent flag = off. A box whose dict was just restarted therefore collects nothing until
|
|
319
|
+
-- the API re-pushes, which is the safe direction: no data beats data nobody asked to pay
|
|
320
|
+
-- for. The DB is the source of truth and re-pushes on every route apply.
|
|
321
|
+
local collect_paths = rules_dict and rules_dict:get("cfg:" .. host .. ":paths")
|
|
322
|
+
|
|
323
|
+
pcall(function()
|
|
324
|
+
if collect_paths and not is_static_asset(uri) then
|
|
325
|
+
-- Hoisted: normalize_path is the single most expensive call in this handler (~1.4 us
|
|
326
|
+
-- of string work), so it runs exactly once even though both the counter key and the
|
|
327
|
+
-- index slot need its result.
|
|
328
|
+
local norm = normalize_path(uri)
|
|
329
|
+
local path_key = gpfx .. "p:" .. norm
|
|
330
|
+
-- incr-then-add: the hot path is ONE hash hit on an existing key. `incr` with
|
|
331
|
+
-- no init returns nil for an absent key, which doubles as the existence probe,
|
|
332
|
+
-- so only a path unseen today reaches the cap check below.
|
|
333
|
+
--
|
|
334
|
+
-- Deliberately no init_ttl here: shdict rejects init_ttl unless init is also
|
|
335
|
+
-- given ("'init_ttl' must be used with 'init'"), and that raises a Lua error
|
|
336
|
+
-- which aborts the whole log handler — taking the status counters below down
|
|
337
|
+
-- with it. The TTL is set by the safe_add on the creating request instead.
|
|
338
|
+
if not analytics:incr(path_key, 1) then
|
|
339
|
+
local distinct = analytics:incr(gpfx .. "pn", 1, 0, D48H) or 0
|
|
340
|
+
if distinct > PATH_CARDINALITY_CAP then
|
|
341
|
+
analytics:incr(gpfx .. "p:other", 1, 0, D48H)
|
|
342
|
+
else
|
|
343
|
+
-- `pn` was already the distinct-path counter, so the slots just give it an
|
|
344
|
+
-- enumerable side. Stored WITHOUT the "p:" prefix; the reader adds it back.
|
|
345
|
+
if analytics:safe_add(path_key, 1, D48H) then
|
|
346
|
+
analytics:set(gpfx .. "pi:" .. distinct, norm, D48H)
|
|
347
|
+
end
|
|
348
|
+
end
|
|
151
349
|
end
|
|
152
350
|
end
|
|
351
|
+
end)
|
|
153
352
|
|
|
154
|
-
--
|
|
353
|
+
-- 3d. Status-code mix. Bounded by construction (a few dozen codes).
|
|
354
|
+
pcall(function()
|
|
355
|
+
bump_indexed(gpfx .. "s:" .. status, D48H, gpfx .. "sn", gpfx .. "si:", status)
|
|
356
|
+
end)
|
|
357
|
+
|
|
358
|
+
-- ── 4. Raw request ring buffer (also the live SSE source) ────────────────────
|
|
359
|
+
--
|
|
360
|
+
-- `seq` is claimed BEFORE encoding so the stored JSON can carry a deterministic
|
|
361
|
+
-- id (`{host}:{seq}`). That id is the dedup key everywhere downstream: /logs/recent
|
|
362
|
+
-- returns these entries verbatim, and pipe_stream.lua streams the same slots live,
|
|
363
|
+
-- so the same request reaches the client with one stable id instead of a fresh
|
|
364
|
+
-- random one per source. `incr` with init 0 never returns nil here; guard anyway
|
|
365
|
+
-- so a shdict hiccup can't index a nil into the slot key.
|
|
155
366
|
|
|
156
367
|
pcall(function()
|
|
368
|
+
local seq = request_data:incr("rlog:" .. host .. ":seq", 1, 0)
|
|
369
|
+
if not seq then return end
|
|
157
370
|
local ok_j, j = pcall(cjson.encode, {
|
|
371
|
+
id = host .. ":" .. seq,
|
|
158
372
|
ip = ip,
|
|
373
|
+
-- nil is simply omitted by cjson, so a box with no GeoIP writes the same shape
|
|
374
|
+
-- minus this key rather than a null the reader has to special-case.
|
|
375
|
+
country = cc,
|
|
159
376
|
ts = ts,
|
|
160
377
|
method = method,
|
|
161
378
|
status = status,
|
|
@@ -166,17 +383,6 @@ pcall(function()
|
|
|
166
383
|
rt = rt,
|
|
167
384
|
})
|
|
168
385
|
if ok_j and j then
|
|
169
|
-
|
|
170
|
-
local slot = seq % RING
|
|
171
|
-
request_data:set("rlog:" .. host .. ":" .. slot, j, D1H)
|
|
386
|
+
request_data:set("rlog:" .. host .. ":" .. (seq % RING), j, D1H)
|
|
172
387
|
end
|
|
173
388
|
end)
|
|
174
|
-
|
|
175
|
-
-- ── 5. Live-log pipe (only when a subscriber is watching) ───────────────────
|
|
176
|
-
-- Fire via timer to avoid blocking the log phase
|
|
177
|
-
|
|
178
|
-
if pipe_log and pipe_log.pipe_request_log
|
|
179
|
-
and request_data:get("log_pipe:sub:" .. host) then
|
|
180
|
-
ngx.timer.at(0, pipe_log.pipe_request_log,
|
|
181
|
-
host, ip, ts, ua, uri, req_len, bytes, rt, method, status)
|
|
182
|
-
end
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
-- Proof that a routing TARGET is ours, as Openship Cloud's shared edge now requires
|
|
2
|
+
-- before forwarding `<slug>.opsh.io` to it.
|
|
3
|
+
--
|
|
4
|
+
-- Keyed on (organization_id, target) because that is what the upstream keys it on:
|
|
5
|
+
-- one target serves every free domain on that box, so this is target-level state,
|
|
6
|
+
-- not domain-level. One install holds several rows — its own address plus one per
|
|
7
|
+
-- remote server it deploys to.
|
|
8
|
+
--
|
|
9
|
+
-- The token is persisted HERE, not only on the edge, because the upstream's list
|
|
10
|
+
-- endpoint returns id/target/status/expiry but NOT the token. Verification lasts 90
|
|
11
|
+
-- days and the upstream re-probes THE SAME token within 7 days of expiry; a 404 then
|
|
12
|
+
-- expires it and the free domain stops resolving ~83 days after a green deploy with
|
|
13
|
+
-- nothing in our logs. So a token that lived only on a box's disk would be
|
|
14
|
+
-- unrecoverable after a rebuild, and rows here are never deleted when a free domain
|
|
15
|
+
-- is dropped.
|
|
16
|
+
--
|
|
17
|
+
-- `server_id` is ON DELETE SET NULL, not cascade: losing the server row must not
|
|
18
|
+
-- take the token with it.
|
|
19
|
+
CREATE TABLE IF NOT EXISTS "edge_target_verification" (
|
|
20
|
+
"id" text PRIMARY KEY NOT NULL,
|
|
21
|
+
"organization_id" text NOT NULL REFERENCES "organization"("id") ON DELETE CASCADE,
|
|
22
|
+
"target" text NOT NULL,
|
|
23
|
+
"host" text NOT NULL,
|
|
24
|
+
"server_id" text REFERENCES "servers"("id") ON DELETE SET NULL,
|
|
25
|
+
"verification_id" integer,
|
|
26
|
+
"token" text,
|
|
27
|
+
"retired_tokens" jsonb,
|
|
28
|
+
"challenge_path" text,
|
|
29
|
+
"status" text NOT NULL DEFAULT 'pending',
|
|
30
|
+
"validated_ip" text,
|
|
31
|
+
"expires_at" timestamp,
|
|
32
|
+
"last_checked_at" timestamp,
|
|
33
|
+
"last_error" text,
|
|
34
|
+
"created_at" timestamp NOT NULL DEFAULT now(),
|
|
35
|
+
"updated_at" timestamp NOT NULL DEFAULT now()
|
|
36
|
+
);
|
|
37
|
+
--> statement-breakpoint
|
|
38
|
+
CREATE UNIQUE INDEX IF NOT EXISTS "uq_edge_target_verification"
|
|
39
|
+
ON "edge_target_verification" ("organization_id", "target");
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
-- Continuous container health monitoring: the durable memory behind the alerts.
|
|
2
|
+
--
|
|
3
|
+
-- Openship knew a container had crashed at exactly ONE moment — the ~15s post-deploy
|
|
4
|
+
-- stabilization watch. After that nobody looked again, so an OOM-kill at 3am or a
|
|
5
|
+
-- Postgres that starts bouncing after a host reboot reached the operator only when a
|
|
6
|
+
-- user complained. The health watch closes that, but a poller that alerts on "the
|
|
7
|
+
-- container isn't running" gets muted within a day: a redeploy recreating containers,
|
|
8
|
+
-- an operator's `docker stop`, an unreachable host, and a crash loop all look exactly
|
|
9
|
+
-- like a failure at a point in time.
|
|
10
|
+
--
|
|
11
|
+
-- So the unit of alerting is an INCIDENT, not a state reading: opened once, escalated
|
|
12
|
+
-- only when it gets worse, resolved once with a downtime duration. This table is that
|
|
13
|
+
-- memory — it survives the control-plane restart that in-process state would not (a box
|
|
14
|
+
-- down for three days must not re-page on every API restart), and it doubles as the
|
|
15
|
+
-- project Health tab's history.
|
|
16
|
+
CREATE TABLE IF NOT EXISTS "service_incident" (
|
|
17
|
+
"id" text PRIMARY KEY NOT NULL,
|
|
18
|
+
"organization_id" text NOT NULL REFERENCES "organization"("id") ON DELETE CASCADE,
|
|
19
|
+
-- Null for SERVER-scoped incidents (an unreachable box gets ONE incident, not one
|
|
20
|
+
-- per service on it — that fan-out is the single loudest source of alert fatigue).
|
|
21
|
+
"project_id" text REFERENCES "project"("id") ON DELETE CASCADE,
|
|
22
|
+
-- The resolved service row, when there is one. SET NULL, not cascade: deleting a
|
|
23
|
+
-- service must not erase the record of the outage it had (`service_key` still
|
|
24
|
+
-- identifies it, and `service_name` is snapshotted below).
|
|
25
|
+
"service_id" text REFERENCES "service"("id") ON DELETE SET NULL,
|
|
26
|
+
-- Stable identity across a container RECREATE, which is the whole point: every
|
|
27
|
+
-- deploy mints a new container id, so keying incidents on container id would open a
|
|
28
|
+
-- fresh incident for the same workload after each redeploy. Service id when we
|
|
29
|
+
-- resolved one, else the container name (single-app deploys have no service row);
|
|
30
|
+
-- `server:<id>` for server-scoped rows.
|
|
31
|
+
"service_key" text NOT NULL,
|
|
32
|
+
-- Snapshot of the display name at incident time — the alert text must still read
|
|
33
|
+
-- correctly after a rename, and history rows must not silently change meaning.
|
|
34
|
+
"service_name" text NOT NULL,
|
|
35
|
+
"server_id" text REFERENCES "servers"("id") ON DELETE SET NULL,
|
|
36
|
+
"container_id" text,
|
|
37
|
+
-- down | crash_loop | unhealthy | server_unreachable. Ordered by severity in the
|
|
38
|
+
-- watcher: an incident escalates upward (and re-notifies), never downward.
|
|
39
|
+
"kind" text NOT NULL,
|
|
40
|
+
"status" text NOT NULL DEFAULT 'open', -- open | resolved
|
|
41
|
+
"reason" text,
|
|
42
|
+
"exit_code" integer,
|
|
43
|
+
"restart_count" integer NOT NULL DEFAULT 0,
|
|
44
|
+
"oom_killed" boolean NOT NULL DEFAULT false,
|
|
45
|
+
-- Consecutive ticks that agreed on this verdict. An incident is only opened (and
|
|
46
|
+
-- only notified) at >= 2, so a snapshot taken mid-restart or during a slow
|
|
47
|
+
-- healthcheck start never pages anyone.
|
|
48
|
+
"confirmations" integer NOT NULL DEFAULT 0,
|
|
49
|
+
-- Notification bookkeeping. Deliberately a count, not a flag: open, escalation and
|
|
50
|
+
-- resolve each notify exactly once, and nothing else ever does.
|
|
51
|
+
"notify_count" integer NOT NULL DEFAULT 0,
|
|
52
|
+
"notified_at" timestamp,
|
|
53
|
+
"log_excerpt" text,
|
|
54
|
+
"opened_at" timestamp NOT NULL DEFAULT now(),
|
|
55
|
+
"resolved_at" timestamp,
|
|
56
|
+
-- Last tick that actually OBSERVED this workload. Not advanced when the host is
|
|
57
|
+
-- unreachable — during a connectivity loss we know nothing, so nothing is claimed.
|
|
58
|
+
"last_seen_at" timestamp NOT NULL DEFAULT now(),
|
|
59
|
+
"created_at" timestamp NOT NULL DEFAULT now(),
|
|
60
|
+
"updated_at" timestamp NOT NULL DEFAULT now()
|
|
61
|
+
);
|
|
62
|
+
--> statement-breakpoint
|
|
63
|
+
-- One open incident per workload — the dedup primitive. As a partial unique index,
|
|
64
|
+
-- "we are already alerting about this" is a DB invariant rather than sweep
|
|
65
|
+
-- bookkeeping: two overlapping ticks (a manual Run now landing on a cron tick) cannot
|
|
66
|
+
-- both open one and double-notify.
|
|
67
|
+
CREATE UNIQUE INDEX IF NOT EXISTS "uq_service_incident_open"
|
|
68
|
+
ON "service_incident" ("project_id", "service_key")
|
|
69
|
+
WHERE "status" = 'open' AND "project_id" IS NOT NULL;
|
|
70
|
+
--> statement-breakpoint
|
|
71
|
+
-- Same invariant for server-scoped rows. Needs its own index because Postgres treats
|
|
72
|
+
-- NULLs as distinct in a unique index, so the one above would happily allow a second
|
|
73
|
+
-- open row for the same box.
|
|
74
|
+
CREATE UNIQUE INDEX IF NOT EXISTS "uq_service_incident_open_server"
|
|
75
|
+
ON "service_incident" ("server_id")
|
|
76
|
+
WHERE "status" = 'open' AND "project_id" IS NULL;
|
|
77
|
+
--> statement-breakpoint
|
|
78
|
+
CREATE INDEX IF NOT EXISTS "idx_service_incident_project"
|
|
79
|
+
ON "service_incident" ("project_id", "opened_at" DESC);
|
|
80
|
+
--> statement-breakpoint
|
|
81
|
+
-- Intent marker. `disableProject` stopped the container and recorded NOTHING, so
|
|
82
|
+
-- "the operator turned this off" was unknowable — the health watch would page about
|
|
83
|
+
-- every deliberately-disabled project, forever. Set on disable, cleared on enable.
|
|
84
|
+
ALTER TABLE "project" ADD COLUMN IF NOT EXISTS "disabled_at" timestamp;
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
-- Widen the daily analytics rollup: distinct visitors, top paths, status mix.
|
|
2
|
+
--
|
|
3
|
+
-- The visitor pipeline already persisted per-country counts, and the dashboard
|
|
4
|
+
-- already rendered cards for "unique IPs" and a top-paths table. Neither was real:
|
|
5
|
+
-- `top_paths` was hardcoded `[]` in the service layer, and "unique IPs" was
|
|
6
|
+
-- `unique_requests` — the count of non-static REQUESTS — so five page views from
|
|
7
|
+
-- one browser read as five visitors. These columns are what makes those two
|
|
8
|
+
-- numbers mean what the UI has been claiming.
|
|
9
|
+
--
|
|
10
|
+
-- All three land HERE, on the daily table, rather than on the per-minute one:
|
|
11
|
+
-- visitors and paths are the high-cardinality metrics, and holding them per-minute
|
|
12
|
+
-- would multiply the edge's shared-dict key count by ~1440 to feed a chart nobody
|
|
13
|
+
-- asked for — evicting, in the process, the request counters they annotate.
|
|
14
|
+
-- Countries already live here for the same reason, and the edge keeps all four
|
|
15
|
+
-- under one `g:{domain}:{day}:` key prefix so a reader gets the whole day in a
|
|
16
|
+
-- single dict scan.
|
|
17
|
+
--
|
|
18
|
+
-- Additive with defaults on purpose: `migrations-additive.test.ts` rejects a
|
|
19
|
+
-- NOT NULL column without a DEFAULT, because an older cross-version dump omits the
|
|
20
|
+
-- column entirely and Drizzle then emits DEFAULT for it — with no default that's a
|
|
21
|
+
-- NULL into NOT NULL on the newer receiver, i.e. a broken cloud/project transfer.
|
|
22
|
+
|
|
23
|
+
-- Distinct visitors for the day, deduplicated AT THE EDGE and stored as a count.
|
|
24
|
+
--
|
|
25
|
+
-- Privacy is a property of where the dedup happens: the edge hashes each address
|
|
26
|
+
-- with a salt that is generated on that box, rotates daily, and never leaves its
|
|
27
|
+
-- shared memory. Only the cardinality of that set is written here. No address and
|
|
28
|
+
-- no per-visitor row exists at any layer, which is what keeps the published
|
|
29
|
+
-- "we don't collect behavioural analytics on your end users" true while still
|
|
30
|
+
-- reporting a real visitor number.
|
|
31
|
+
--
|
|
32
|
+
-- Understates on a very busy domain: the edge's `visitors` zone LRU-evicts past
|
|
33
|
+
-- roughly 1M distinct/day. mgmt_api `GET /status` reports that zone's free space so
|
|
34
|
+
-- a reader can label the number approximate instead of presenting an eviction
|
|
35
|
+
-- artifact as a measurement.
|
|
36
|
+
ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "visitors" integer DEFAULT 0 NOT NULL;
|
|
37
|
+
--> statement-breakpoint
|
|
38
|
+
|
|
39
|
+
-- Top paths: { "/": 900, "/orders/:id": 120, "other": 40 }.
|
|
40
|
+
--
|
|
41
|
+
-- Normalized at the edge before it ever becomes a key: query string stripped (so a
|
|
42
|
+
-- ?token= or ?session= can never be persisted here), numeric and UUID segments
|
|
43
|
+
-- collapsed to `:id`, key length capped, and the tail past a cardinality cap folded
|
|
44
|
+
-- into "other" — otherwise a scanner walking /wp-admin variants mints thousands of
|
|
45
|
+
-- permanent keys and evicts the real counters.
|
|
46
|
+
ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "paths" jsonb;
|
|
47
|
+
--> statement-breakpoint
|
|
48
|
+
|
|
49
|
+
-- Status-code mix: { "200": 4210, "404": 17, "502": 3 }. Status was captured into
|
|
50
|
+
-- the edge's raw-request ring buffer (RAM, 1h) but never aggregated, so error rate
|
|
51
|
+
-- was unanswerable from any persisted data. Bounded by construction.
|
|
52
|
+
ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "statuses" jsonb;
|
|
53
|
+
--> statement-breakpoint
|
|
54
|
+
|
|
55
|
+
-- Widen the per-minute bandwidth counters from int4 to int8.
|
|
56
|
+
--
|
|
57
|
+
-- These are bytes for ONE MINUTE, and int4 caps at 2,147,483,647 — about 2.1 GB,
|
|
58
|
+
-- which a single minute reaches at roughly 286 Mbps. That is unremarkable for a
|
|
59
|
+
-- site serving video or large downloads. Past it Postgres raises "integer out of
|
|
60
|
+
-- range", the upsert for that minute fails, and the scrape for the whole domain
|
|
61
|
+
-- dies with it — so the busiest domains on a box would be precisely the ones with
|
|
62
|
+
-- no analytics, and it would look like the feature simply didn't work for them.
|
|
63
|
+
--
|
|
64
|
+
-- Now that collection runs unattended on a schedule rather than only when someone
|
|
65
|
+
-- opens the tab, that failure would happen in a job with nobody reading the error.
|
|
66
|
+
--
|
|
67
|
+
-- Widening is lossless and there is no application change: the edge already
|
|
68
|
+
-- accumulates these as Lua numbers (doubles, exact to 2^53) and Drizzle reads them
|
|
69
|
+
-- back with mode:"number", which is exact over the same range.
|
|
70
|
+
ALTER TABLE "server_analytics" ALTER COLUMN "bandwidth_in" TYPE bigint;
|
|
71
|
+
--> statement-breakpoint
|
|
72
|
+
ALTER TABLE "server_analytics" ALTER COLUMN "bandwidth_out" TYPE bigint;
|