openship 0.5.0 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/README.md +26 -1
  2. package/dist/bare-IWS2QHJ3.js +23 -0
  3. package/dist/{bowser-O3AEA6AN.js → bowser-HNOTPGJO.js} +2 -2
  4. package/dist/{chunk-5MZ624VX.js → chunk-2GWYYHOR.js} +1 -1
  5. package/dist/{chunk-BS2AXYZX.js → chunk-2NZSUXOB.js} +647 -214
  6. package/dist/chunk-3FYIXVX5.js +8051 -0
  7. package/dist/{chunk-ILFETZOM.js → chunk-3LDWWY7U.js} +21 -1
  8. package/dist/{chunk-W73X43R7.js → chunk-3NQQXMDR.js} +3 -3
  9. package/dist/{chunk-NNPEG6HM.js → chunk-3NZICQHF.js} +2 -2
  10. package/dist/{chunk-OWED2GRM.js → chunk-42WQABXM.js} +2 -2
  11. package/dist/{chunk-TYVME3XD.js → chunk-6FH4AXVY.js} +20 -3
  12. package/dist/{chunk-ZCOQJCZO.js → chunk-6ODUBKUB.js} +3 -3
  13. package/dist/{chunk-DYPH4HCI.js → chunk-AO252K7Z.js} +1 -1
  14. package/dist/{chunk-IXCHURVK.js → chunk-DOD2BXHQ.js} +4 -4
  15. package/dist/{chunk-VWPEKOH6.js → chunk-DRK3MEDS.js} +3 -3
  16. package/dist/{chunk-V7BZMGOT.js → chunk-EMXI4UCU.js} +5 -5
  17. package/dist/{chunk-RB35YN3J.js → chunk-EW6KB2J7.js} +391 -63
  18. package/dist/chunk-I3DPMTSF.js +29 -0
  19. package/dist/{chunk-NXFIRK2Z.js → chunk-IOT5ZX23.js} +573 -459
  20. package/dist/{chunk-KZOBYJUW.js → chunk-IW2UKBVF.js} +3 -3
  21. package/dist/{chunk-YWMODKWL.js → chunk-IXM5245F.js} +165 -13
  22. package/dist/{chunk-KNTJ7PSF.js → chunk-JIMUYEKN.js} +4 -4
  23. package/dist/{chunk-O6ZZF52V.js → chunk-ONKEYSMY.js} +2 -2
  24. package/dist/{chunk-UMQP7SYT.js → chunk-OTLSJYAC.js} +1316 -53
  25. package/dist/{chunk-EI575GHN.js → chunk-PDW4L3NV.js} +3 -3
  26. package/dist/{chunk-KI2EY3WB.js → chunk-PXKCZB2I.js} +1 -1
  27. package/dist/{chunk-6APIPIJY.js → chunk-QWHBRY2K.js} +5 -5
  28. package/dist/{chunk-JLKKYUZA.js → chunk-RRJJGVKE.js} +7 -7
  29. package/dist/{chunk-6B5YCFFV.js → chunk-RVER2EOO.js} +5 -5
  30. package/dist/{chunk-Q2CIRAOK.js → chunk-RXQBD4ZY.js} +1 -1
  31. package/dist/{chunk-6LKHXB6T.js → chunk-TWJCLD4W.js} +48 -13
  32. package/dist/{chunk-K2RPVGPN.js → chunk-UC7YIEHT.js} +36 -17
  33. package/dist/{chunk-I6PSTQIR.js → chunk-VDDMADOF.js} +3 -3
  34. package/dist/{chunk-6PIYRG5I.js → chunk-VZ2ZHPS2.js} +52 -42
  35. package/dist/{chunk-AVIB7CT7.js → chunk-WNWQBB7F.js} +2 -2
  36. package/dist/{chunk-FRNAR33O.js → chunk-XZYFZ6A5.js} +3 -3
  37. package/dist/chunk-YIPMDGG6.js +703 -0
  38. package/dist/{chunk-RPHRPFEH.js → chunk-ZCFVHA25.js} +1 -1
  39. package/dist/{chunk-FPRHYBY2.js → chunk-ZFJXPCQP.js} +1 -1
  40. package/dist/{chunk-JYGD5WSN.js → chunk-ZWLDLTJP.js} +1 -1
  41. package/dist/{cloud-7QYOGF7F.js → cloud-KRTZ2IL5.js} +3 -3
  42. package/dist/cloud-M2YLWEXO.js +31 -0
  43. package/dist/{detect-BAJMY6KM.js → detect-ALF4GXRX.js} +7 -8
  44. package/dist/{dist-IQJ6H6UZ.js → dist-BM6NPWIH.js} +3 -3
  45. package/dist/{dist-es-X3B5QYKB.js → dist-es-2YRDXMUU.js} +6 -6
  46. package/dist/{dist-es-OSNJBLXB.js → dist-es-CCVZEFRZ.js} +5 -5
  47. package/dist/{dist-es-VNKHQVIB.js → dist-es-IOH6G4GM.js} +4 -4
  48. package/dist/{dist-es-RYHIPOTN.js → dist-es-JGBPABJI.js} +14 -14
  49. package/dist/{dist-es-JSTDUS7Q.js → dist-es-KWSMDPIP.js} +6 -6
  50. package/dist/{dist-es-5DY7K32L.js → dist-es-UYDDOVZ3.js} +6 -6
  51. package/dist/{dist-es-WV4AMOII.js → dist-es-XUOR4NQN.js} +8 -8
  52. package/dist/docker-QECQJ5P4.js +32 -0
  53. package/dist/docker-edge-executor-62N64MGS.js +17 -0
  54. package/dist/edge-import-SAYTRJLO.js +48 -0
  55. package/dist/ensure-container-edge-CFSCPGZK.js +32 -0
  56. package/dist/{event-streams-EHRAKL75.js → event-streams-TR4GAC53.js} +4 -4
  57. package/dist/{executor-LOYMPU67.js → executor-PT6UEJ5M.js} +8 -8
  58. package/dist/index.js +1228 -463
  59. package/dist/{loadSso-46FQRIZS.js → loadSso-UGF4JMRK.js} +10 -10
  60. package/dist/{local-executor-V7U5VOZW.js → local-executor-WOFZD72X.js} +4 -4
  61. package/dist/nginx-YOEEFBTE.js +31 -0
  62. package/dist/node-entry.js +4 -0
  63. package/dist/{noop-FMMFG2EZ.js → noop-NJ72F46T.js} +3 -3
  64. package/dist/openresty-lua-WGBO2M3Z.js +81 -0
  65. package/dist/server/index.js +23276 -8015
  66. package/dist/server/lua/geo_country.lua +34 -8
  67. package/dist/server/lua/maxminddb.lua +403 -0
  68. package/dist/server/lua/mgmt_api.lua +431 -68
  69. package/dist/server/lua/pipe_stream.lua +52 -39
  70. package/dist/server/lua/site_logger.lua +236 -30
  71. package/dist/server/migrations/0082_edge_target_verification.sql +39 -0
  72. package/dist/server/migrations/0083_service_incident.sql +84 -0
  73. package/dist/server/migrations/0084_analytics_daily_rollup.sql +72 -0
  74. package/dist/server/migrations/0085_resource_usage.sql +68 -0
  75. package/dist/server/migrations/0086_remote_infra_updates.sql +41 -0
  76. package/dist/server/migrations/0087_server_container_version_columns.sql +17 -0
  77. package/dist/server/migrations/0088_audit_source_and_settings.sql +27 -0
  78. package/dist/server/migrations/0089_auto_scan_infra.sql +7 -0
  79. package/dist/server/migrations/0090_project_collect_paths.sql +18 -0
  80. package/dist/server/migrations/0091_project_server_id.sql +26 -0
  81. package/dist/server/migrations/0092_project_internal_alias.sql +13 -0
  82. package/dist/server/migrations/0093_service_incident_org_index.sql +13 -0
  83. package/dist/server/migrations/0094_backup_restore_meta.sql +15 -0
  84. package/dist/server/migrations/0095_backup_restore_cancel.sql +19 -0
  85. package/dist/server/migrations/0096_backup_policy_retention_defaults.sql +20 -0
  86. package/dist/server/migrations/0097_update_status_upstream_only.sql +35 -0
  87. package/dist/server/migrations/meta/_journal.json +112 -0
  88. package/dist/setup-M2ZDGH3O.js +21 -0
  89. package/dist/{signin-ITOPKKUB.js → signin-3M3UPB6K.js} +10 -10
  90. package/dist/{sso-oidc-NON2PD7Y.js → sso-oidc-3VDIYOEJ.js} +10 -10
  91. package/dist/{sts-HHFFSAY2.js → sts-B3WIYLXR.js} +9 -9
  92. package/package.json +6 -3
  93. package/dist/bare-VRGE5MXH.js +0 -23
  94. package/dist/chunk-CUJXGK7R.js +0 -3167
  95. package/dist/chunk-EAAGSQXN.js +0 -377
  96. package/dist/cloud-3M2WZ624.js +0 -30
  97. package/dist/docker-2NGBRJFJ.js +0 -18
  98. package/dist/docker-edge-executor-5I4M5QT7.js +0 -13
  99. package/dist/edge-import-RIWNEIXQ.js +0 -47
  100. package/dist/ensure-container-edge-3NSYKHRX.js +0 -31
  101. package/dist/nginx-QE4DOM5J.js +0 -24
  102. package/dist/openresty-lua-BTAKECCF.js +0 -42
  103. package/dist/server/lua/pipe_log.lua +0 -71
  104. package/dist/setup-KELALK7H.js +0 -20
@@ -2,6 +2,14 @@
2
2
  -- content_by_lua: SSE endpoint for real-time request log streaming.
3
3
  -- GET /logs/stream?domain=example.com
4
4
  -- Internal only - 127.0.0.1:9145
5
+ --
6
+ -- Reads the SAME ring buffer site_logger.lua writes (rlog:{domain}:seq +
7
+ -- rlog:{domain}:{slot}), by a PER-CONNECTION cursor. Nothing is consumed: N
8
+ -- concurrent watchers each walk the ring independently and none steals another's
9
+ -- frames. This is what replaced the old single-subscriber log_pipe:q queue —
10
+ -- a containerized edge tails this over `docker exec curl`, and a curl that
11
+ -- outlives its client (the daemon buffers its stdout, so no EPIPE) can no longer
12
+ -- drain the queue out from under the next reader.
5
13
 
6
14
  local sh = ngx.shared.request_data
7
15
  if not sh then
@@ -20,12 +28,10 @@ end
20
28
  domain = domain:lower()
21
29
  if domain:sub(1, 4) == "www." then domain = domain:sub(5) end
22
30
 
23
- local SUB_KEY = "log_pipe:sub:" .. domain
24
- local QUEUE_KEY = "log_pipe:q:" .. domain
25
-
26
- -- Clear stale queue entries, mark subscriber active
27
- sh:delete(QUEUE_KEY)
28
- sh:set(SUB_KEY, true, 30)
31
+ -- MUST match site_logger.lua's RING — same slot arithmetic on both ends.
32
+ local RING = 1000
33
+ local SEQ_KEY = "rlog:" .. domain .. ":seq"
34
+ local slot_key = function(seq) return "rlog:" .. domain .. ":" .. (seq % RING) end
29
35
 
30
36
  ngx.header["Content-Type"] = "text/event-stream"
31
37
  ngx.header["Cache-Control"] = "no-cache, no-store"
@@ -36,55 +42,62 @@ ngx.header["X-Accel-Buffering"] = "no"
36
42
  -- Without this, ngx.flush on an empty buffer sends chunked-EOF (0\r\n\r\n)
37
43
  -- and terminates the response immediately.
38
44
  ngx.print(": connected\n\n")
39
- if not ngx.flush(true) then
40
- sh:delete(SUB_KEY)
41
- return
42
- end
45
+ if not ngx.flush(true) then return end
43
46
 
44
- local started = ngx.now()
45
- local last_hb = started
46
- local last_ref = started
47
+ -- Start at the current head: stream only requests that arrive AFTER connect.
48
+ -- History up to this point is served separately by /logs/recent, and the client
49
+ -- dedups the two by the deterministic `{host}:{seq}` id, so a row seen by both
50
+ -- collapses to one.
51
+ local cursor = tonumber(sh:get(SEQ_KEY)) or 0
52
+ local started = ngx.now()
53
+ local last_hb = started
47
54
 
48
55
  while true do
49
- -- Max 1 hour per connection
50
- if ngx.now() - started > 3600 then
51
- sh:delete(SUB_KEY)
52
- return
53
- end
56
+ -- Max 1 hour per connection.
57
+ if ngx.now() - started > 3600 then return end
58
+
59
+ local head = tonumber(sh:get(SEQ_KEY)) or 0
60
+
61
+ -- If we've fallen more than a full ring behind, the oldest unread slots have
62
+ -- already been overwritten. Jump to the oldest slot still holding its own
63
+ -- entry (head-RING maps to head's slot, so head-RING+1 is the oldest live one)
64
+ -- and skip the lost range — same bound the old 2000-entry queue had.
65
+ if cursor < head - RING then cursor = head - RING end
54
66
 
55
- -- Drain up to 100 queued entries per cycle
56
67
  local sent = 0
57
- while sent < 100 do
58
- local entry = sh:rpop(QUEUE_KEY)
59
- if not entry then break end
60
- ngx.print("event: request\ndata: ", entry, "\n\n")
61
- sent = sent + 1
68
+ while cursor < head and sent < 100 do
69
+ local next_seq = cursor + 1
70
+ local entry = sh:get(slot_key(next_seq))
71
+ if entry then
72
+ ngx.print("event: request\ndata: ", entry, "\n\n")
73
+ sent = sent + 1
74
+ cursor = next_seq
75
+ elseif next_seq < head then
76
+ -- A later seq already exists, so this can't be the incr(seq)→set(slot)
77
+ -- write window: the slot is genuinely lost (expired or wrapped). Skip it.
78
+ cursor = next_seq
79
+ else
80
+ -- next_seq == head: the newest entry's slot may not be written yet
81
+ -- (the microsecond window between incr(seq) and set(slot)). Retry it
82
+ -- next cycle without advancing.
83
+ break
84
+ end
62
85
  end
63
86
 
64
87
  if sent > 0 then
65
- if not ngx.flush(true) then
66
- sh:delete(SUB_KEY)
67
- return
68
- end
88
+ if not ngx.flush(true) then return end
69
89
  end
70
90
 
71
91
  local now = ngx.now()
72
92
 
73
- -- Heartbeat every 15s
93
+ -- Heartbeat every 15s. Its flush is also how a dead client is detected: once
94
+ -- the reader (a `docker exec curl`) goes away and its socket closes, this
95
+ -- flush fails and the connection tears down here.
74
96
  if now - last_hb > 15 then
75
97
  ngx.print(": ping\n\n")
76
- if not ngx.flush(true) then
77
- sh:delete(SUB_KEY)
78
- return
79
- end
98
+ if not ngx.flush(true) then return end
80
99
  last_hb = now
81
100
  end
82
101
 
83
- -- Refresh subscriber TTL every 10s
84
- if now - last_ref > 10 then
85
- sh:set(SUB_KEY, true, 30)
86
- last_ref = now
87
- end
88
-
89
102
  ngx.sleep(0.05)
90
103
  end
@@ -8,17 +8,33 @@
8
8
  -- s:{domain}:{epoch_min}:i bandwidth in bytes (TTL 24h)
9
9
  -- s:{domain}:{epoch_min}:o bandwidth out bytes (TTL 24h)
10
10
  -- s:{domain}:{epoch_min}:t response time sum (seconds) (TTL 24h)
11
- -- s:{domain}:{epoch_min}:u unique (non-static) reqs (TTL 24h)
12
- -- g:{domain}:{YYYYMMDD}:{CC} country hit count (TTL 48h)
11
+ -- s:{domain}:{epoch_min}:u page (non-static) requests (TTL 24h)
13
12
  -- c:{domain}:{epoch_min}:{CC} country per minute (TTL 24h)
14
13
  -- t:{domain}:r / :i / :o lifetime totals (no TTL)
15
14
  -- d:{domain} domain index marker (no TTL)
16
15
  --
16
+ -- Daily rollup, all under one `g:{domain}:{YYYYMMDD}:` prefix so mgmt_api's
17
+ -- /analytics/geo serves the whole day in a SINGLE dict scan:
18
+ -- g:{domain}:{YYYYMMDD}:{CC} country hit count (TTL 48h)
19
+ -- g:{domain}:{YYYYMMDD}:v distinct visitors (TTL 48h)
20
+ -- g:{domain}:{YYYYMMDD}:p:{path} path hit count (TTL 48h)
21
+ -- g:{domain}:{YYYYMMDD}:s:{code} status-code count (TTL 48h)
22
+ --
23
+ -- Shared dict key schema (visitors zone):
24
+ -- vsalt:{YYYYMMDD} per-day hash salt (TTL 48h)
25
+ -- vd:{domain}:{YYYYMMDD}:{hash} distinct-visitor marker (TTL 48h)
26
+ --
17
27
  -- Shared dict key schema (request_data zone):
18
28
  -- rlog:{domain}:seq monotonic write pointer
19
29
  -- rlog:{domain}:{slot} JSON entry (ring buf) (TTL 1h)
20
- -- log_pipe:sub:{domain} live subscriber flag (TTL 30s)
21
- -- log_pipe:q:{domain} live log queue entries
30
+ --
31
+ -- The ring is ALSO the live source: pipe_stream.lua's SSE endpoint reads these
32
+ -- slots by a per-connection cursor, so there is no separate live queue or
33
+ -- subscriber flag — every watcher sees the full stream and steals from none.
34
+ --
35
+ -- NOTE on `:u` — it counts non-static REQUESTS, not people. It was surfaced as
36
+ -- "unique IPs" all the way to the dashboard, which it never was. Real distinct
37
+ -- visitors are the `vd:`/`:v` pair below.
22
38
 
23
39
  local cjson = require "cjson.safe"
24
40
 
@@ -29,12 +45,17 @@ if not geo_ok then
29
45
  ngx.log(ngx.WARN, "[site_logger] geo_country module not available. GeoIP disabled.")
30
46
  end
31
47
 
32
- -- SAFE LOAD: Pipe module (per-worker, cached by require)
33
- local pipe_ok, pipe_log = pcall(require, "openship.pipe_log")
34
- if not pipe_ok then pipe_log = nil end
35
-
36
48
  local analytics = ngx.shared.analytics
37
49
  local request_data = ngx.shared.request_data
50
+ -- Per-host edge config, pushed by the API (see mgmt_api's /analytics/config).
51
+ --
52
+ -- The `rules` zone, NOT `analytics`: this holds a handful of small config values and is
53
+ -- never under eviction pressure, whereas the analytics zone is a 256 MB churn of counters.
54
+ -- An evicted flag there would silently stop path collection with nothing to explain why.
55
+ local rules_dict = ngx.shared.rules
56
+ -- Optional: a box whose nginx.conf predates this zone keeps working, just without
57
+ -- distinct-visitor counts. Never assume a dict exists.
58
+ local visitors = ngx.shared.visitors
38
59
  if not analytics or not request_data then return end
39
60
 
40
61
  -- ── Helpers ──────────────────────────────────────────────────────────────────
@@ -89,11 +110,71 @@ local function is_static_asset(u)
89
110
  return false
90
111
  end
91
112
 
113
+ -- Collapse a request URI into a low-cardinality bucket suitable as a dict key.
114
+ -- Without this, "top paths" is unbounded: every ?token=, every /orders/48219, and
115
+ -- every 404 a scanner probes mints a permanent key, and the analytics zone
116
+ -- LRU-evicts real counters to store junk.
117
+ local function normalize_path(u)
118
+ -- Query string carries ids/tokens and never identifies the route.
119
+ local path = u:match("^([^?#]*)") or u
120
+ if path == "" then return "/" end
121
+ -- Numeric and UUID/hex segments are ids, not routes: /orders/48219 and
122
+ -- /orders/48220 are one path.
123
+ path = path:gsub("/%d+", "/:id")
124
+ path = path:gsub("/%x%x%x%x%x%x%x%x%-?%x*%-?%x*%-?%x*%-?%x*", "/:id")
125
+ -- Bound the key itself — a long URI must not become a long key.
126
+ if #path > 120 then path = path:sub(1, 120) .. "…" end
127
+ return path
128
+ end
129
+
130
+ -- ── Enumerable sets, without a zone scan ─────────────────────────────────────
131
+ --
132
+ -- Reading these counters back used to need key DISCOVERY, and a shared dict offers exactly
133
+ -- one mechanism for that: `get_keys`, which walks EVERY key in the zone holding its lock.
134
+ -- Cost scaled with how much was in the 256 MB zone, not with what was being asked for, and
135
+ -- it ran on the mgmt path four different ways.
136
+ --
137
+ -- So every unbounded set now carries its own index: a counter plus numbered slots.
138
+ --
139
+ -- <counter> how many distinct members
140
+ -- <prefix><n> the n-th member
141
+ --
142
+ -- A reader reads the counter and then reads slots 1..n. Bounded by the answer.
143
+ --
144
+ -- The write only happens on a member's FIRST sighting, so steady state costs nothing: the
145
+ -- millionth request from Germany today takes the plain `incr` path and touches no index.
146
+ local function index_add(counter_key, slot_prefix, member, ttl)
147
+ local n = analytics:incr(counter_key, 1, 0, ttl)
148
+ if n then analytics:set(slot_prefix .. n, member, ttl) end
149
+ end
150
+
151
+ --- Add 1 to a counter, creating it (with TTL) and indexing it on first sighting.
152
+ --
153
+ -- `incr` with no `init` returns nil for an absent key, which doubles as the existence
154
+ -- probe — so the common case is ONE hash operation. TTL is set by the creating `safe_add`
155
+ -- rather than passed to `incr`: shdict rejects `init_ttl` without `init`, and that raises a
156
+ -- Lua error which aborts the whole log handler (it has silently zeroed every counter below
157
+ -- it before).
158
+ local function bump_indexed(key, ttl, counter_key, slot_prefix, member)
159
+ if analytics:incr(key, 1) then return end
160
+ if analytics:safe_add(key, 1, ttl) then
161
+ index_add(counter_key, slot_prefix, member, ttl)
162
+ else
163
+ -- Another worker created it between our probe and our add; our +1 still counts.
164
+ analytics:incr(key, 1)
165
+ end
166
+ end
167
+
92
168
  local RING = 1000
93
169
  local D24H = 86400
94
170
  local D48H = 172800
95
171
  local D1H = 3600
96
172
 
173
+ -- Distinct paths tracked per domain per day. Past this the tail folds into
174
+ -- "other" rather than growing the zone — a URL-fuzzing bot is capped at one
175
+ -- extra key, not thousands.
176
+ local PATH_CARDINALITY_CAP = 2000
177
+
97
178
  -- ── Capture ──────────────────────────────────────────────────────────────────
98
179
 
99
180
  local host = normalize(ngx.var.host)
@@ -109,6 +190,23 @@ local rt = tonumber(ngx.var.request_time) or 0 -- seconds (float)
109
190
  local method = ngx.var.request_method or "GET"
110
191
  local status = ngx.var.status or "0"
111
192
 
193
+ -- Client country, resolved ONCE per request.
194
+ --
195
+ -- Hoisted out of the geo block because both the daily/per-minute rollups and the raw
196
+ -- ring buffer need it. The ring is the single source for the request log — /logs/recent
197
+ -- reads it directly and the live SSE stream (pipe_stream.lua) reads the same slots by
198
+ -- cursor — so resolving the country here means every row carries its flag no matter
199
+ -- which path delivered it. An earlier split (looked up in the rollup, again in a
200
+ -- separate live pipe) left the ring with no country at all.
201
+ --
202
+ -- pcall'd because a corrupt or partially-written mmdb makes the lookup raise, and an
203
+ -- error here would abort every counter below it (see the section header).
204
+ local cc = nil
205
+ if geo and geo.get_country_code then
206
+ local ok_cc, res = pcall(geo.get_country_code, ip)
207
+ if ok_cc and type(res) == "string" and res ~= "" then cc = res end
208
+ end
209
+
112
210
  -- ── 1. Minute-bucket counters ────────────────────────────────────────────────
113
211
 
114
212
  local minute = math.floor(ts / 60)
@@ -136,26 +234,145 @@ analytics:incr("t:" .. host .. ":r", 1, 0)
136
234
  analytics:incr("t:" .. host .. ":i", req_len, 0)
137
235
  analytics:incr("t:" .. host .. ":o", bytes, 0)
138
236
 
139
- -- Domain index (set once, ignore subsequent "exists" errors)
140
- analytics:safe_add("d:" .. host, 1)
237
+ -- Domain index. `safe_add` returning true means this is the first request this box has
238
+ -- ever seen for the host, which is exactly when the enumerable index needs appending —
239
+ -- so domain discovery no longer costs a zone scan either.
240
+ if analytics:safe_add("d:" .. host, 1) then
241
+ index_add("dn", "di:", host)
242
+ end
141
243
 
142
- -- ── 3. GeoIP ─────────────────────────────────────────────────────────────────
244
+ -- ── 3. Daily rollup: geo, visitors, paths, statuses ──────────────────────────
245
+ --
246
+ -- Each sub-block is pcall'd SEPARATELY, and that is load-bearing rather than
247
+ -- defensive habit. A raised error in log_by_lua aborts the remainder of the
248
+ -- handler, so one bad shdict call silently zeroes every counter written after it —
249
+ -- which is exactly what an `incr(..., nil, ttl)` here did: paths AND statuses read
250
+ -- as "no traffic" while requests and bandwidth looked perfect. Isolating them
251
+ -- means a future edit can lose at most its own metric.
252
+
253
+ local day = today()
254
+ local gpfx = "g:" .. host .. ":" .. day .. ":"
143
255
 
144
- if geo then
145
- local cc = geo.get_country_code(ip)
146
- if cc and cc ~= "" then
256
+ -- 3a. GeoIP. `cc` was resolved once above and is reused by the ring buffer and the
257
+ -- live pipe, so a request costs exactly one mmdb lookup no matter how many consumers
258
+ -- report its country.
259
+ if cc then
260
+ pcall(function()
147
261
  -- Daily geo (for /analytics/geo endpoint)
148
- analytics:incr("g:" .. host .. ":" .. today() .. ":" .. cc, 1, 0, D48H)
149
- -- Per-minute geo (for time-series country breakdown)
150
- analytics:incr("c:" .. host .. ":" .. minute .. ":" .. cc, 1, 0, D24H)
262
+ bump_indexed(gpfx .. cc, D48H, gpfx .. "cn", gpfx .. "ci:", cc)
263
+ -- Per-minute geo (for the time-series country breakdown). Indexed per MINUTE, so
264
+ -- the scraper reads only the minutes in its window.
265
+ local mpfx = "c:" .. host .. ":" .. minute .. ":"
266
+ bump_indexed(mpfx .. cc, D24H, mpfx .. "n", mpfx .. "i:", cc)
267
+ end)
268
+ end
269
+
270
+ -- 3b. Distinct visitors — a COUNT, never an identity.
271
+ --
272
+ -- The address is hashed with a salt that is generated on this box, rotates daily,
273
+ -- and is never written anywhere but this shared dict, so the marker keys are not
274
+ -- reversible to an IP and are worthless the next day. Only the resulting counter
275
+ -- is ever flushed to Postgres; no per-visitor row exists at any layer. That is
276
+ -- what lets us report real visitor numbers while still not collecting behavioural
277
+ -- analytics on anyone's end users.
278
+ --
279
+ -- Per DAY, not per minute: a per-minute set is ~1440x the keys for a number
280
+ -- nobody reads, and would evict the counters it exists to annotate.
281
+ pcall(function()
282
+ if visitors then
283
+ local salt_key = "vsalt:" .. day
284
+ local salt = visitors:get(salt_key)
285
+ if not salt then
286
+ -- First writer wins; a racing worker's safe_add fails and it re-reads. The
287
+ -- salt only has to be unpredictable and stable for the day.
288
+ salt = tostring(ngx.now()) .. tostring(math.random(1, 2147483647))
289
+ local ok = visitors:safe_add(salt_key, salt, D48H)
290
+ if not ok then salt = visitors:get(salt_key) or salt end
291
+ end
292
+ -- safe_add is the dedup: it succeeds ONLY on this visitor's first request
293
+ -- today, so the counter below increments exactly once per distinct visitor.
294
+ local marker = "vd:" .. host .. ":" .. day .. ":" .. ngx.crc32_long(salt .. ip)
295
+ if visitors:safe_add(marker, 1, D48H) then
296
+ analytics:incr(gpfx .. "v", 1, 0, D48H)
297
+ end
298
+ end
299
+ end)
300
+
301
+ -- 3c. Top paths — OPT-IN, non-static only, normalized and cardinality-capped.
302
+ --
303
+ -- Off unless the project turned it on, because this one block is the most expensive thing
304
+ -- the log handler does. Measured on the shipped edge image, per request:
305
+ --
306
+ -- paths (whole block) 1.72 us <- 57% of the counter path
307
+ -- of which string work 1.38 us normalize_path + is_static_asset
308
+ -- minute buckets (5 incr) 0.61 us
309
+ -- country (2 incr) 0.24 us
310
+ -- status (1 incr) 0.14 us
311
+ -- this flag read (1 get) 0.07 us
312
+ --
313
+ -- So the gate costs 0.07 us and saves 1.72 us — the counter path drops from ~3.0 us to
314
+ -- ~1.35 us with paths off. It is also the highest-cardinality dimension (up to
315
+ -- PATH_CARDINALITY_CAP keys per domain per day, against ~200 countries and a few dozen
316
+ -- statuses) and the largest column in the daily rollup.
317
+ --
318
+ -- Absent flag = off. A box whose dict was just restarted therefore collects nothing until
319
+ -- the API re-pushes, which is the safe direction: no data beats data nobody asked to pay
320
+ -- for. The DB is the source of truth and re-pushes on every route apply.
321
+ local collect_paths = rules_dict and rules_dict:get("cfg:" .. host .. ":paths")
322
+
323
+ pcall(function()
324
+ if collect_paths and not is_static_asset(uri) then
325
+ -- Hoisted: normalize_path is the single most expensive call in this handler (~1.4 us
326
+ -- of string work), so it runs exactly once even though both the counter key and the
327
+ -- index slot need its result.
328
+ local norm = normalize_path(uri)
329
+ local path_key = gpfx .. "p:" .. norm
330
+ -- incr-then-add: the hot path is ONE hash hit on an existing key. `incr` with
331
+ -- no init returns nil for an absent key, which doubles as the existence probe,
332
+ -- so only a path unseen today reaches the cap check below.
333
+ --
334
+ -- Deliberately no init_ttl here: shdict rejects init_ttl unless init is also
335
+ -- given ("'init_ttl' must be used with 'init'"), and that raises a Lua error
336
+ -- which aborts the whole log handler — taking the status counters below down
337
+ -- with it. The TTL is set by the safe_add on the creating request instead.
338
+ if not analytics:incr(path_key, 1) then
339
+ local distinct = analytics:incr(gpfx .. "pn", 1, 0, D48H) or 0
340
+ if distinct > PATH_CARDINALITY_CAP then
341
+ analytics:incr(gpfx .. "p:other", 1, 0, D48H)
342
+ else
343
+ -- `pn` was already the distinct-path counter, so the slots just give it an
344
+ -- enumerable side. Stored WITHOUT the "p:" prefix; the reader adds it back.
345
+ if analytics:safe_add(path_key, 1, D48H) then
346
+ analytics:set(gpfx .. "pi:" .. distinct, norm, D48H)
347
+ end
348
+ end
151
349
  end
152
350
  end
351
+ end)
153
352
 
154
- -- ── 4. Raw request ring buffer ──────────────────────────────────────────────
353
+ -- 3d. Status-code mix. Bounded by construction (a few dozen codes).
354
+ pcall(function()
355
+ bump_indexed(gpfx .. "s:" .. status, D48H, gpfx .. "sn", gpfx .. "si:", status)
356
+ end)
357
+
358
+ -- ── 4. Raw request ring buffer (also the live SSE source) ────────────────────
359
+ --
360
+ -- `seq` is claimed BEFORE encoding so the stored JSON can carry a deterministic
361
+ -- id (`{host}:{seq}`). That id is the dedup key everywhere downstream: /logs/recent
362
+ -- returns these entries verbatim, and pipe_stream.lua streams the same slots live,
363
+ -- so the same request reaches the client with one stable id instead of a fresh
364
+ -- random one per source. `incr` with init 0 never returns nil here; guard anyway
365
+ -- so a shdict hiccup can't index a nil into the slot key.
155
366
 
156
367
  pcall(function()
368
+ local seq = request_data:incr("rlog:" .. host .. ":seq", 1, 0)
369
+ if not seq then return end
157
370
  local ok_j, j = pcall(cjson.encode, {
371
+ id = host .. ":" .. seq,
158
372
  ip = ip,
373
+ -- nil is simply omitted by cjson, so a box with no GeoIP writes the same shape
374
+ -- minus this key rather than a null the reader has to special-case.
375
+ country = cc,
159
376
  ts = ts,
160
377
  method = method,
161
378
  status = status,
@@ -166,17 +383,6 @@ pcall(function()
166
383
  rt = rt,
167
384
  })
168
385
  if ok_j and j then
169
- local seq = request_data:incr("rlog:" .. host .. ":seq", 1, 0)
170
- local slot = seq % RING
171
- request_data:set("rlog:" .. host .. ":" .. slot, j, D1H)
386
+ request_data:set("rlog:" .. host .. ":" .. (seq % RING), j, D1H)
172
387
  end
173
388
  end)
174
-
175
- -- ── 5. Live-log pipe (only when a subscriber is watching) ───────────────────
176
- -- Fire via timer to avoid blocking the log phase
177
-
178
- if pipe_log and pipe_log.pipe_request_log
179
- and request_data:get("log_pipe:sub:" .. host) then
180
- ngx.timer.at(0, pipe_log.pipe_request_log,
181
- host, ip, ts, ua, uri, req_len, bytes, rt, method, status)
182
- end
@@ -0,0 +1,39 @@
1
+ -- Proof that a routing TARGET is ours, as Openship Cloud's shared edge now requires
2
+ -- before forwarding `<slug>.opsh.io` to it.
3
+ --
4
+ -- Keyed on (organization_id, target) because that is what the upstream keys it on:
5
+ -- one target serves every free domain on that box, so this is target-level state,
6
+ -- not domain-level. One install holds several rows — its own address plus one per
7
+ -- remote server it deploys to.
8
+ --
9
+ -- The token is persisted HERE, not only on the edge, because the upstream's list
10
+ -- endpoint returns id/target/status/expiry but NOT the token. Verification lasts 90
11
+ -- days and the upstream re-probes THE SAME token within 7 days of expiry; a 404 then
12
+ -- expires it and the free domain stops resolving ~83 days after a green deploy with
13
+ -- nothing in our logs. So a token that lived only on a box's disk would be
14
+ -- unrecoverable after a rebuild, and rows here are never deleted when a free domain
15
+ -- is dropped.
16
+ --
17
+ -- `server_id` is ON DELETE SET NULL, not cascade: losing the server row must not
18
+ -- take the token with it.
19
+ CREATE TABLE IF NOT EXISTS "edge_target_verification" (
20
+ "id" text PRIMARY KEY NOT NULL,
21
+ "organization_id" text NOT NULL REFERENCES "organization"("id") ON DELETE CASCADE,
22
+ "target" text NOT NULL,
23
+ "host" text NOT NULL,
24
+ "server_id" text REFERENCES "servers"("id") ON DELETE SET NULL,
25
+ "verification_id" integer,
26
+ "token" text,
27
+ "retired_tokens" jsonb,
28
+ "challenge_path" text,
29
+ "status" text NOT NULL DEFAULT 'pending',
30
+ "validated_ip" text,
31
+ "expires_at" timestamp,
32
+ "last_checked_at" timestamp,
33
+ "last_error" text,
34
+ "created_at" timestamp NOT NULL DEFAULT now(),
35
+ "updated_at" timestamp NOT NULL DEFAULT now()
36
+ );
37
+ --> statement-breakpoint
38
+ CREATE UNIQUE INDEX IF NOT EXISTS "uq_edge_target_verification"
39
+ ON "edge_target_verification" ("organization_id", "target");
@@ -0,0 +1,84 @@
1
+ -- Continuous container health monitoring: the durable memory behind the alerts.
2
+ --
3
+ -- Openship knew a container had crashed at exactly ONE moment — the ~15s post-deploy
4
+ -- stabilization watch. After that nobody looked again, so an OOM-kill at 3am or a
5
+ -- Postgres that starts bouncing after a host reboot reached the operator only when a
6
+ -- user complained. The health watch closes that, but a poller that alerts on "the
7
+ -- container isn't running" gets muted within a day: a redeploy recreating containers,
8
+ -- an operator's `docker stop`, an unreachable host, and a crash loop all look exactly
9
+ -- like a failure at a point in time.
10
+ --
11
+ -- So the unit of alerting is an INCIDENT, not a state reading: opened once, escalated
12
+ -- only when it gets worse, resolved once with a downtime duration. This table is that
13
+ -- memory — it survives the control-plane restart that in-process state would not (a box
14
+ -- down for three days must not re-page on every API restart), and it doubles as the
15
+ -- project Health tab's history.
16
+ CREATE TABLE IF NOT EXISTS "service_incident" (
17
+ "id" text PRIMARY KEY NOT NULL,
18
+ "organization_id" text NOT NULL REFERENCES "organization"("id") ON DELETE CASCADE,
19
+ -- Null for SERVER-scoped incidents (an unreachable box gets ONE incident, not one
20
+ -- per service on it — that fan-out is the single loudest source of alert fatigue).
21
+ "project_id" text REFERENCES "project"("id") ON DELETE CASCADE,
22
+ -- The resolved service row, when there is one. SET NULL, not cascade: deleting a
23
+ -- service must not erase the record of the outage it had (`service_key` still
24
+ -- identifies it, and `service_name` is snapshotted below).
25
+ "service_id" text REFERENCES "service"("id") ON DELETE SET NULL,
26
+ -- Stable identity across a container RECREATE, which is the whole point: every
27
+ -- deploy mints a new container id, so keying incidents on container id would open a
28
+ -- fresh incident for the same workload after each redeploy. Service id when we
29
+ -- resolved one, else the container name (single-app deploys have no service row);
30
+ -- `server:<id>` for server-scoped rows.
31
+ "service_key" text NOT NULL,
32
+ -- Snapshot of the display name at incident time — the alert text must still read
33
+ -- correctly after a rename, and history rows must not silently change meaning.
34
+ "service_name" text NOT NULL,
35
+ "server_id" text REFERENCES "servers"("id") ON DELETE SET NULL,
36
+ "container_id" text,
37
+ -- down | crash_loop | unhealthy | server_unreachable. Ordered by severity in the
38
+ -- watcher: an incident escalates upward (and re-notifies), never downward.
39
+ "kind" text NOT NULL,
40
+ "status" text NOT NULL DEFAULT 'open', -- open | resolved
41
+ "reason" text,
42
+ "exit_code" integer,
43
+ "restart_count" integer NOT NULL DEFAULT 0,
44
+ "oom_killed" boolean NOT NULL DEFAULT false,
45
+ -- Consecutive ticks that agreed on this verdict. An incident is only opened (and
46
+ -- only notified) at >= 2, so a snapshot taken mid-restart or during a slow
47
+ -- healthcheck start never pages anyone.
48
+ "confirmations" integer NOT NULL DEFAULT 0,
49
+ -- Notification bookkeeping. Deliberately a count, not a flag: open, escalation and
50
+ -- resolve each notify exactly once, and nothing else ever does.
51
+ "notify_count" integer NOT NULL DEFAULT 0,
52
+ "notified_at" timestamp,
53
+ "log_excerpt" text,
54
+ "opened_at" timestamp NOT NULL DEFAULT now(),
55
+ "resolved_at" timestamp,
56
+ -- Last tick that actually OBSERVED this workload. Not advanced when the host is
57
+ -- unreachable — during a connectivity loss we know nothing, so nothing is claimed.
58
+ "last_seen_at" timestamp NOT NULL DEFAULT now(),
59
+ "created_at" timestamp NOT NULL DEFAULT now(),
60
+ "updated_at" timestamp NOT NULL DEFAULT now()
61
+ );
62
+ --> statement-breakpoint
63
+ -- One open incident per workload — the dedup primitive. As a partial unique index,
64
+ -- "we are already alerting about this" is a DB invariant rather than sweep
65
+ -- bookkeeping: two overlapping ticks (a manual Run now landing on a cron tick) cannot
66
+ -- both open one and double-notify.
67
+ CREATE UNIQUE INDEX IF NOT EXISTS "uq_service_incident_open"
68
+ ON "service_incident" ("project_id", "service_key")
69
+ WHERE "status" = 'open' AND "project_id" IS NOT NULL;
70
+ --> statement-breakpoint
71
+ -- Same invariant for server-scoped rows. Needs its own index because Postgres treats
72
+ -- NULLs as distinct in a unique index, so the one above would happily allow a second
73
+ -- open row for the same box.
74
+ CREATE UNIQUE INDEX IF NOT EXISTS "uq_service_incident_open_server"
75
+ ON "service_incident" ("server_id")
76
+ WHERE "status" = 'open' AND "project_id" IS NULL;
77
+ --> statement-breakpoint
78
+ CREATE INDEX IF NOT EXISTS "idx_service_incident_project"
79
+ ON "service_incident" ("project_id", "opened_at" DESC);
80
+ --> statement-breakpoint
81
+ -- Intent marker. `disableProject` stopped the container and recorded NOTHING, so
82
+ -- "the operator turned this off" was unknowable — the health watch would page about
83
+ -- every deliberately-disabled project, forever. Set on disable, cleared on enable.
84
+ ALTER TABLE "project" ADD COLUMN IF NOT EXISTS "disabled_at" timestamp;
@@ -0,0 +1,72 @@
1
+ -- Widen the daily analytics rollup: distinct visitors, top paths, status mix.
2
+ --
3
+ -- The visitor pipeline already persisted per-country counts, and the dashboard
4
+ -- already rendered cards for "unique IPs" and a top-paths table. Neither was real:
5
+ -- `top_paths` was hardcoded `[]` in the service layer, and "unique IPs" was
6
+ -- `unique_requests` — the count of non-static REQUESTS — so five page views from
7
+ -- one browser read as five visitors. These columns are what makes those two
8
+ -- numbers mean what the UI has been claiming.
9
+ --
10
+ -- All three land HERE, on the daily table, rather than on the per-minute one:
11
+ -- visitors and paths are the high-cardinality metrics, and holding them per-minute
12
+ -- would multiply the edge's shared-dict key count by ~1440 to feed a chart nobody
13
+ -- asked for — evicting, in the process, the request counters they annotate.
14
+ -- Countries already live here for the same reason, and the edge keeps all four
15
+ -- under one `g:{domain}:{day}:` key prefix so a reader gets the whole day in a
16
+ -- single dict scan.
17
+ --
18
+ -- Additive with defaults on purpose: `migrations-additive.test.ts` rejects a
19
+ -- NOT NULL column without a DEFAULT, because an older cross-version dump omits the
20
+ -- column entirely and Drizzle then emits DEFAULT for it — with no default that's a
21
+ -- NULL into NOT NULL on the newer receiver, i.e. a broken cloud/project transfer.
22
+
23
+ -- Distinct visitors for the day, deduplicated AT THE EDGE and stored as a count.
24
+ --
25
+ -- Privacy is a property of where the dedup happens: the edge hashes each address
26
+ -- with a salt that is generated on that box, rotates daily, and never leaves its
27
+ -- shared memory. Only the cardinality of that set is written here. No address and
28
+ -- no per-visitor row exists at any layer, which is what keeps the published
29
+ -- "we don't collect behavioural analytics on your end users" true while still
30
+ -- reporting a real visitor number.
31
+ --
32
+ -- Understates on a very busy domain: the edge's `visitors` zone LRU-evicts past
33
+ -- roughly 1M distinct/day. mgmt_api `GET /status` reports that zone's free space so
34
+ -- a reader can label the number approximate instead of presenting an eviction
35
+ -- artifact as a measurement.
36
+ ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "visitors" integer DEFAULT 0 NOT NULL;
37
+ --> statement-breakpoint
38
+
39
+ -- Top paths: { "/": 900, "/orders/:id": 120, "other": 40 }.
40
+ --
41
+ -- Normalized at the edge before it ever becomes a key: query string stripped (so a
42
+ -- ?token= or ?session= can never be persisted here), numeric and UUID segments
43
+ -- collapsed to `:id`, key length capped, and the tail past a cardinality cap folded
44
+ -- into "other" — otherwise a scanner walking /wp-admin variants mints thousands of
45
+ -- permanent keys and evicts the real counters.
46
+ ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "paths" jsonb;
47
+ --> statement-breakpoint
48
+
49
+ -- Status-code mix: { "200": 4210, "404": 17, "502": 3 }. Status was captured into
50
+ -- the edge's raw-request ring buffer (RAM, 1h) but never aggregated, so error rate
51
+ -- was unanswerable from any persisted data. Bounded by construction.
52
+ ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "statuses" jsonb;
53
+ --> statement-breakpoint
54
+
55
+ -- Widen the per-minute bandwidth counters from int4 to int8.
56
+ --
57
+ -- These are bytes for ONE MINUTE, and int4 caps at 2,147,483,647 — about 2.1 GB,
58
+ -- which a single minute reaches at roughly 286 Mbps. That is unremarkable for a
59
+ -- site serving video or large downloads. Past it Postgres raises "integer out of
60
+ -- range", the upsert for that minute fails, and the scrape for the whole domain
61
+ -- dies with it — so the busiest domains on a box would be precisely the ones with
62
+ -- no analytics, and it would look like the feature simply didn't work for them.
63
+ --
64
+ -- Now that collection runs unattended on a schedule rather than only when someone
65
+ -- opens the tab, that failure would happen in a job with nobody reading the error.
66
+ --
67
+ -- Widening is lossless and there is no application change: the edge already
68
+ -- accumulates these as Lua numbers (doubles, exact to 2^53) and Drizzle reads them
69
+ -- back with mode:"number", which is exact over the same range.
70
+ ALTER TABLE "server_analytics" ALTER COLUMN "bandwidth_in" TYPE bigint;
71
+ --> statement-breakpoint
72
+ ALTER TABLE "server_analytics" ALTER COLUMN "bandwidth_out" TYPE bigint;