openship 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/README.md +26 -1
  2. package/dist/bare-FSUAM5V5.js +23 -0
  3. package/dist/{bowser-O3AEA6AN.js → bowser-HNOTPGJO.js} +2 -2
  4. package/dist/{chunk-5MZ624VX.js → chunk-2GWYYHOR.js} +1 -1
  5. package/dist/chunk-3FYIXVX5.js +8051 -0
  6. package/dist/{chunk-ILFETZOM.js → chunk-3LDWWY7U.js} +21 -1
  7. package/dist/{chunk-W73X43R7.js → chunk-3NQQXMDR.js} +3 -3
  8. package/dist/{chunk-OWED2GRM.js → chunk-42WQABXM.js} +2 -2
  9. package/dist/{chunk-TYVME3XD.js → chunk-6FH4AXVY.js} +20 -3
  10. package/dist/{chunk-ZCOQJCZO.js → chunk-6ODUBKUB.js} +3 -3
  11. package/dist/{chunk-DYPH4HCI.js → chunk-AO252K7Z.js} +1 -1
  12. package/dist/{chunk-K2RPVGPN.js → chunk-B3CE2OMR.js} +14 -12
  13. package/dist/{chunk-IXCHURVK.js → chunk-DOD2BXHQ.js} +4 -4
  14. package/dist/{chunk-VWPEKOH6.js → chunk-DRK3MEDS.js} +3 -3
  15. package/dist/{chunk-V7BZMGOT.js → chunk-EMXI4UCU.js} +5 -5
  16. package/dist/{chunk-NNPEG6HM.js → chunk-EP2TNALB.js} +2 -2
  17. package/dist/{chunk-NXFIRK2Z.js → chunk-GUQ3QCJ7.js} +573 -459
  18. package/dist/chunk-I3DPMTSF.js +29 -0
  19. package/dist/{chunk-KZOBYJUW.js → chunk-IW2UKBVF.js} +3 -3
  20. package/dist/{chunk-YWMODKWL.js → chunk-IXM5245F.js} +165 -13
  21. package/dist/{chunk-KNTJ7PSF.js → chunk-JIMUYEKN.js} +4 -4
  22. package/dist/chunk-OGY63DGD.js +705 -0
  23. package/dist/{chunk-O6ZZF52V.js → chunk-ONKEYSMY.js} +2 -2
  24. package/dist/{chunk-UMQP7SYT.js → chunk-OTLSJYAC.js} +1316 -53
  25. package/dist/{chunk-EI575GHN.js → chunk-PDW4L3NV.js} +3 -3
  26. package/dist/{chunk-JLKKYUZA.js → chunk-PSZSKEPA.js} +7 -7
  27. package/dist/{chunk-KI2EY3WB.js → chunk-PXKCZB2I.js} +1 -1
  28. package/dist/{chunk-RB35YN3J.js → chunk-QEH4QUBX.js} +391 -63
  29. package/dist/{chunk-6APIPIJY.js → chunk-QWHBRY2K.js} +5 -5
  30. package/dist/{chunk-6B5YCFFV.js → chunk-RVER2EOO.js} +5 -5
  31. package/dist/{chunk-Q2CIRAOK.js → chunk-RXQBD4ZY.js} +1 -1
  32. package/dist/{chunk-6LKHXB6T.js → chunk-TWJCLD4W.js} +48 -13
  33. package/dist/{chunk-I6PSTQIR.js → chunk-VDDMADOF.js} +3 -3
  34. package/dist/{chunk-BS2AXYZX.js → chunk-VE2YOK4I.js} +626 -214
  35. package/dist/{chunk-6PIYRG5I.js → chunk-VZ2ZHPS2.js} +52 -42
  36. package/dist/{chunk-AVIB7CT7.js → chunk-WNWQBB7F.js} +2 -2
  37. package/dist/{chunk-FRNAR33O.js → chunk-XZYFZ6A5.js} +3 -3
  38. package/dist/{chunk-RPHRPFEH.js → chunk-ZCFVHA25.js} +1 -1
  39. package/dist/{chunk-FPRHYBY2.js → chunk-ZFJXPCQP.js} +1 -1
  40. package/dist/{chunk-JYGD5WSN.js → chunk-ZWLDLTJP.js} +1 -1
  41. package/dist/cloud-E5M6VMCG.js +31 -0
  42. package/dist/{cloud-7QYOGF7F.js → cloud-KRTZ2IL5.js} +3 -3
  43. package/dist/{detect-BAJMY6KM.js → detect-ALF4GXRX.js} +7 -8
  44. package/dist/{dist-IQJ6H6UZ.js → dist-BM6NPWIH.js} +3 -3
  45. package/dist/{dist-es-X3B5QYKB.js → dist-es-2YRDXMUU.js} +6 -6
  46. package/dist/{dist-es-OSNJBLXB.js → dist-es-CCVZEFRZ.js} +5 -5
  47. package/dist/{dist-es-VNKHQVIB.js → dist-es-IOH6G4GM.js} +4 -4
  48. package/dist/{dist-es-RYHIPOTN.js → dist-es-JGBPABJI.js} +14 -14
  49. package/dist/{dist-es-JSTDUS7Q.js → dist-es-KWSMDPIP.js} +6 -6
  50. package/dist/{dist-es-5DY7K32L.js → dist-es-UYDDOVZ3.js} +6 -6
  51. package/dist/{dist-es-WV4AMOII.js → dist-es-XUOR4NQN.js} +8 -8
  52. package/dist/docker-QECQJ5P4.js +32 -0
  53. package/dist/docker-edge-executor-62N64MGS.js +17 -0
  54. package/dist/edge-import-CQWHN2Q3.js +48 -0
  55. package/dist/ensure-container-edge-MJVXJQG5.js +32 -0
  56. package/dist/{event-streams-EHRAKL75.js → event-streams-TR4GAC53.js} +4 -4
  57. package/dist/{executor-LOYMPU67.js → executor-JELARVKC.js} +8 -8
  58. package/dist/index.js +1170 -459
  59. package/dist/{loadSso-46FQRIZS.js → loadSso-UGF4JMRK.js} +10 -10
  60. package/dist/{local-executor-V7U5VOZW.js → local-executor-WOFZD72X.js} +4 -4
  61. package/dist/nginx-AXGNWRKZ.js +31 -0
  62. package/dist/node-entry.js +4 -0
  63. package/dist/{noop-FMMFG2EZ.js → noop-NJ72F46T.js} +3 -3
  64. package/dist/openresty-lua-IE4P5YQX.js +81 -0
  65. package/dist/server/index.js +23172 -7994
  66. package/dist/server/lua/geo_country.lua +31 -6
  67. package/dist/server/lua/maxminddb.lua +403 -0
  68. package/dist/server/lua/mgmt_api.lua +431 -68
  69. package/dist/server/lua/pipe_log.lua +10 -4
  70. package/dist/server/lua/site_logger.lua +220 -12
  71. package/dist/server/migrations/0082_edge_target_verification.sql +39 -0
  72. package/dist/server/migrations/0083_service_incident.sql +84 -0
  73. package/dist/server/migrations/0084_analytics_daily_rollup.sql +72 -0
  74. package/dist/server/migrations/0085_resource_usage.sql +68 -0
  75. package/dist/server/migrations/0086_remote_infra_updates.sql +41 -0
  76. package/dist/server/migrations/0087_server_container_version_columns.sql +17 -0
  77. package/dist/server/migrations/0088_audit_source_and_settings.sql +27 -0
  78. package/dist/server/migrations/0089_auto_scan_infra.sql +7 -0
  79. package/dist/server/migrations/0090_project_collect_paths.sql +18 -0
  80. package/dist/server/migrations/0091_project_server_id.sql +26 -0
  81. package/dist/server/migrations/0092_project_internal_alias.sql +13 -0
  82. package/dist/server/migrations/0093_service_incident_org_index.sql +13 -0
  83. package/dist/server/migrations/0094_backup_restore_meta.sql +15 -0
  84. package/dist/server/migrations/0095_backup_restore_cancel.sql +19 -0
  85. package/dist/server/migrations/0096_backup_policy_retention_defaults.sql +20 -0
  86. package/dist/server/migrations/0097_update_status_upstream_only.sql +35 -0
  87. package/dist/server/migrations/meta/_journal.json +112 -0
  88. package/dist/setup-FA4PW4NU.js +21 -0
  89. package/dist/{signin-ITOPKKUB.js → signin-3M3UPB6K.js} +10 -10
  90. package/dist/{sso-oidc-NON2PD7Y.js → sso-oidc-3VDIYOEJ.js} +10 -10
  91. package/dist/{sts-HHFFSAY2.js → sts-B3WIYLXR.js} +9 -9
  92. package/package.json +6 -3
  93. package/dist/bare-VRGE5MXH.js +0 -23
  94. package/dist/chunk-CUJXGK7R.js +0 -3167
  95. package/dist/chunk-EAAGSQXN.js +0 -377
  96. package/dist/cloud-3M2WZ624.js +0 -30
  97. package/dist/docker-2NGBRJFJ.js +0 -18
  98. package/dist/docker-edge-executor-5I4M5QT7.js +0 -13
  99. package/dist/edge-import-RIWNEIXQ.js +0 -47
  100. package/dist/ensure-container-edge-3NSYKHRX.js +0 -31
  101. package/dist/nginx-QE4DOM5J.js +0 -24
  102. package/dist/openresty-lua-BTAKECCF.js +0 -42
  103. package/dist/setup-KELALK7H.js +0 -20
@@ -8,17 +8,31 @@
8
8
  -- s:{domain}:{epoch_min}:i bandwidth in bytes (TTL 24h)
9
9
  -- s:{domain}:{epoch_min}:o bandwidth out bytes (TTL 24h)
10
10
  -- s:{domain}:{epoch_min}:t response time sum (seconds) (TTL 24h)
11
- -- s:{domain}:{epoch_min}:u unique (non-static) reqs (TTL 24h)
12
- -- g:{domain}:{YYYYMMDD}:{CC} country hit count (TTL 48h)
11
+ -- s:{domain}:{epoch_min}:u page (non-static) requests (TTL 24h)
13
12
  -- c:{domain}:{epoch_min}:{CC} country per minute (TTL 24h)
14
13
  -- t:{domain}:r / :i / :o lifetime totals (no TTL)
15
14
  -- d:{domain} domain index marker (no TTL)
16
15
  --
16
+ -- Daily rollup, all under one `g:{domain}:{YYYYMMDD}:` prefix so mgmt_api's
17
+ -- /analytics/geo serves the whole day in a SINGLE dict scan:
18
+ -- g:{domain}:{YYYYMMDD}:{CC} country hit count (TTL 48h)
19
+ -- g:{domain}:{YYYYMMDD}:v distinct visitors (TTL 48h)
20
+ -- g:{domain}:{YYYYMMDD}:p:{path} path hit count (TTL 48h)
21
+ -- g:{domain}:{YYYYMMDD}:s:{code} status-code count (TTL 48h)
22
+ --
23
+ -- Shared dict key schema (visitors zone):
24
+ -- vsalt:{YYYYMMDD} per-day hash salt (TTL 48h)
25
+ -- vd:{domain}:{YYYYMMDD}:{hash} distinct-visitor marker (TTL 48h)
26
+ --
17
27
  -- Shared dict key schema (request_data zone):
18
28
  -- rlog:{domain}:seq monotonic write pointer
19
29
  -- rlog:{domain}:{slot} JSON entry (ring buf) (TTL 1h)
20
30
  -- log_pipe:sub:{domain} live subscriber flag (TTL 30s)
21
31
  -- log_pipe:q:{domain} live log queue entries
32
+ --
33
+ -- NOTE on `:u` — it counts non-static REQUESTS, not people. It was surfaced as
34
+ -- "unique IPs" all the way to the dashboard, which it never was. Real distinct
35
+ -- visitors are the `vd:`/`:v` pair below.
22
36
 
23
37
  local cjson = require "cjson.safe"
24
38
 
@@ -35,6 +49,15 @@ if not pipe_ok then pipe_log = nil end
35
49
 
36
50
  local analytics = ngx.shared.analytics
37
51
  local request_data = ngx.shared.request_data
52
+ -- Per-host edge config, pushed by the API (see mgmt_api's /analytics/config).
53
+ --
54
+ -- The `rules` zone, NOT `analytics`: this holds a handful of small config values and is
55
+ -- never under eviction pressure, whereas the analytics zone is a 256 MB churn of counters.
56
+ -- An evicted flag there would silently stop path collection with nothing to explain why.
57
+ local rules_dict = ngx.shared.rules
58
+ -- Optional: a box whose nginx.conf predates this zone keeps working, just without
59
+ -- distinct-visitor counts. Never assume a dict exists.
60
+ local visitors = ngx.shared.visitors
38
61
  if not analytics or not request_data then return end
39
62
 
40
63
  -- ── Helpers ──────────────────────────────────────────────────────────────────
@@ -89,11 +112,71 @@ local function is_static_asset(u)
89
112
  return false
90
113
  end
91
114
 
115
+ -- Collapse a request URI into a low-cardinality bucket suitable as a dict key.
116
+ -- Without this, "top paths" is unbounded: every ?token=, every /orders/48219, and
117
+ -- every 404 a scanner probes mints a permanent key, and the analytics zone
118
+ -- LRU-evicts real counters to store junk.
119
+ local function normalize_path(u)
120
+ -- Query string carries ids/tokens and never identifies the route.
121
+ local path = u:match("^([^?#]*)") or u
122
+ if path == "" then return "/" end
123
+ -- Numeric and UUID/hex segments are ids, not routes: /orders/48219 and
124
+ -- /orders/48220 are one path.
125
+ path = path:gsub("/%d+", "/:id")
126
+ path = path:gsub("/%x%x%x%x%x%x%x%x%-?%x*%-?%x*%-?%x*%-?%x*", "/:id")
127
+ -- Bound the key itself — a long URI must not become a long key.
128
+ if #path > 120 then path = path:sub(1, 120) .. "…" end
129
+ return path
130
+ end
131
+
132
+ -- ── Enumerable sets, without a zone scan ─────────────────────────────────────
133
+ --
134
+ -- Reading these counters back used to need key DISCOVERY, and a shared dict offers exactly
135
+ -- one mechanism for that: `get_keys`, which walks EVERY key in the zone holding its lock.
136
+ -- Cost scaled with how much was in the 256 MB zone, not with what was being asked for, and
137
+ -- it ran on the mgmt path four different ways.
138
+ --
139
+ -- So every unbounded set now carries its own index: a counter plus numbered slots.
140
+ --
141
+ -- <counter> how many distinct members
142
+ -- <prefix><n> the n-th member
143
+ --
144
+ -- A reader reads the counter and then reads slots 1..n. Bounded by the answer.
145
+ --
146
+ -- The write only happens on a member's FIRST sighting, so steady state costs nothing: the
147
+ -- millionth request from Germany today takes the plain `incr` path and touches no index.
148
+ local function index_add(counter_key, slot_prefix, member, ttl)
149
+ local n = analytics:incr(counter_key, 1, 0, ttl)
150
+ if n then analytics:set(slot_prefix .. n, member, ttl) end
151
+ end
152
+
153
+ --- Add 1 to a counter, creating it (with TTL) and indexing it on first sighting.
154
+ --
155
+ -- `incr` with no `init` returns nil for an absent key, which doubles as the existence
156
+ -- probe — so the common case is ONE hash operation. TTL is set by the creating `safe_add`
157
+ -- rather than passed to `incr`: shdict rejects `init_ttl` without `init`, and that raises a
158
+ -- Lua error which aborts the whole log handler (it has silently zeroed every counter below
159
+ -- it before).
160
+ local function bump_indexed(key, ttl, counter_key, slot_prefix, member)
161
+ if analytics:incr(key, 1) then return end
162
+ if analytics:safe_add(key, 1, ttl) then
163
+ index_add(counter_key, slot_prefix, member, ttl)
164
+ else
165
+ -- Another worker created it between our probe and our add; our +1 still counts.
166
+ analytics:incr(key, 1)
167
+ end
168
+ end
169
+
92
170
  local RING = 1000
93
171
  local D24H = 86400
94
172
  local D48H = 172800
95
173
  local D1H = 3600
96
174
 
175
+ -- Distinct paths tracked per domain per day. Past this the tail folds into
176
+ -- "other" rather than growing the zone — a URL-fuzzing bot is capped at one
177
+ -- extra key, not thousands.
178
+ local PATH_CARDINALITY_CAP = 2000
179
+
97
180
  -- ── Capture ──────────────────────────────────────────────────────────────────
98
181
 
99
182
  local host = normalize(ngx.var.host)
@@ -109,6 +192,22 @@ local rt = tonumber(ngx.var.request_time) or 0 -- seconds (float)
109
192
  local method = ngx.var.request_method or "GET"
110
193
  local status = ngx.var.status or "0"
111
194
 
195
+ -- Client country, resolved ONCE per request.
196
+ --
197
+ -- Hoisted out of the geo block because THREE consumers need it — the daily/per-minute
198
+ -- rollups, the raw ring buffer behind /logs/recent, and the live SSE pipe. It used to be
199
+ -- looked up inside the rollup and again inside pipe_log, so the ring buffer had no
200
+ -- country at all: a request-log list showed flags on rows that arrived live and none on
201
+ -- rows backfilled from /logs/recent, for the same traffic.
202
+ --
203
+ -- pcall'd because a corrupt or partially-written mmdb makes the lookup raise, and an
204
+ -- error here would abort every counter below it (see the section header).
205
+ local cc = nil
206
+ if geo and geo.get_country_code then
207
+ local ok_cc, res = pcall(geo.get_country_code, ip)
208
+ if ok_cc and type(res) == "string" and res ~= "" then cc = res end
209
+ end
210
+
112
211
  -- ── 1. Minute-bucket counters ────────────────────────────────────────────────
113
212
 
114
213
  local minute = math.floor(ts / 60)
@@ -136,26 +235,135 @@ analytics:incr("t:" .. host .. ":r", 1, 0)
136
235
  analytics:incr("t:" .. host .. ":i", req_len, 0)
137
236
  analytics:incr("t:" .. host .. ":o", bytes, 0)
138
237
 
139
- -- Domain index (set once, ignore subsequent "exists" errors)
140
- analytics:safe_add("d:" .. host, 1)
238
+ -- Domain index. `safe_add` returning true means this is the first request this box has
239
+ -- ever seen for the host, which is exactly when the enumerable index needs appending —
240
+ -- so domain discovery no longer costs a zone scan either.
241
+ if analytics:safe_add("d:" .. host, 1) then
242
+ index_add("dn", "di:", host)
243
+ end
244
+
245
+ -- ── 3. Daily rollup: geo, visitors, paths, statuses ──────────────────────────
246
+ --
247
+ -- Each sub-block is pcall'd SEPARATELY, and that is load-bearing rather than
248
+ -- defensive habit. A raised error in log_by_lua aborts the remainder of the
249
+ -- handler, so one bad shdict call silently zeroes every counter written after it —
250
+ -- which is exactly what an `incr(..., nil, ttl)` here did: paths AND statuses read
251
+ -- as "no traffic" while requests and bandwidth looked perfect. Isolating them
252
+ -- means a future edit can lose at most its own metric.
141
253
 
142
- -- ── 3. GeoIP ─────────────────────────────────────────────────────────────────
254
+ local day = today()
255
+ local gpfx = "g:" .. host .. ":" .. day .. ":"
143
256
 
144
- if geo then
145
- local cc = geo.get_country_code(ip)
146
- if cc and cc ~= "" then
257
+ -- 3a. GeoIP. `cc` was resolved once above and is reused by the ring buffer and the
258
+ -- live pipe, so a request costs exactly one mmdb lookup no matter how many consumers
259
+ -- report its country.
260
+ if cc then
261
+ pcall(function()
147
262
  -- Daily geo (for /analytics/geo endpoint)
148
- analytics:incr("g:" .. host .. ":" .. today() .. ":" .. cc, 1, 0, D48H)
149
- -- Per-minute geo (for time-series country breakdown)
150
- analytics:incr("c:" .. host .. ":" .. minute .. ":" .. cc, 1, 0, D24H)
263
+ bump_indexed(gpfx .. cc, D48H, gpfx .. "cn", gpfx .. "ci:", cc)
264
+ -- Per-minute geo (for the time-series country breakdown). Indexed per MINUTE, so
265
+ -- the scraper reads only the minutes in its window.
266
+ local mpfx = "c:" .. host .. ":" .. minute .. ":"
267
+ bump_indexed(mpfx .. cc, D24H, mpfx .. "n", mpfx .. "i:", cc)
268
+ end)
269
+ end
270
+
271
+ -- 3b. Distinct visitors — a COUNT, never an identity.
272
+ --
273
+ -- The address is hashed with a salt that is generated on this box, rotates daily,
274
+ -- and is never written anywhere but this shared dict, so the marker keys are not
275
+ -- reversible to an IP and are worthless the next day. Only the resulting counter
276
+ -- is ever flushed to Postgres; no per-visitor row exists at any layer. That is
277
+ -- what lets us report real visitor numbers while still not collecting behavioural
278
+ -- analytics on anyone's end users.
279
+ --
280
+ -- Per DAY, not per minute: a per-minute set is ~1440x the keys for a number
281
+ -- nobody reads, and would evict the counters it exists to annotate.
282
+ pcall(function()
283
+ if visitors then
284
+ local salt_key = "vsalt:" .. day
285
+ local salt = visitors:get(salt_key)
286
+ if not salt then
287
+ -- First writer wins; a racing worker's safe_add fails and it re-reads. The
288
+ -- salt only has to be unpredictable and stable for the day.
289
+ salt = tostring(ngx.now()) .. tostring(math.random(1, 2147483647))
290
+ local ok = visitors:safe_add(salt_key, salt, D48H)
291
+ if not ok then salt = visitors:get(salt_key) or salt end
292
+ end
293
+ -- safe_add is the dedup: it succeeds ONLY on this visitor's first request
294
+ -- today, so the counter below increments exactly once per distinct visitor.
295
+ local marker = "vd:" .. host .. ":" .. day .. ":" .. ngx.crc32_long(salt .. ip)
296
+ if visitors:safe_add(marker, 1, D48H) then
297
+ analytics:incr(gpfx .. "v", 1, 0, D48H)
151
298
  end
152
299
  end
300
+ end)
301
+
302
+ -- 3c. Top paths — OPT-IN, non-static only, normalized and cardinality-capped.
303
+ --
304
+ -- Off unless the project turned it on, because this one block is the most expensive thing
305
+ -- the log handler does. Measured on the shipped edge image, per request:
306
+ --
307
+ -- paths (whole block) 1.72 us <- 57% of the counter path
308
+ -- of which string work 1.38 us normalize_path + is_static_asset
309
+ -- minute buckets (5 incr) 0.61 us
310
+ -- country (2 incr) 0.24 us
311
+ -- status (1 incr) 0.14 us
312
+ -- this flag read (1 get) 0.07 us
313
+ --
314
+ -- So the gate costs 0.07 us and saves 1.72 us — the counter path drops from ~3.0 us to
315
+ -- ~1.35 us with paths off. It is also the highest-cardinality dimension (up to
316
+ -- PATH_CARDINALITY_CAP keys per domain per day, against ~200 countries and a few dozen
317
+ -- statuses) and the largest column in the daily rollup.
318
+ --
319
+ -- Absent flag = off. A box whose dict was just restarted therefore collects nothing until
320
+ -- the API re-pushes, which is the safe direction: no data beats data nobody asked to pay
321
+ -- for. The DB is the source of truth and re-pushes on every route apply.
322
+ local collect_paths = rules_dict and rules_dict:get("cfg:" .. host .. ":paths")
323
+
324
+ pcall(function()
325
+ if collect_paths and not is_static_asset(uri) then
326
+ -- Hoisted: normalize_path is the single most expensive call in this handler (~1.4 us
327
+ -- of string work), so it runs exactly once even though both the counter key and the
328
+ -- index slot need its result.
329
+ local norm = normalize_path(uri)
330
+ local path_key = gpfx .. "p:" .. norm
331
+ -- incr-then-add: the hot path is ONE hash hit on an existing key. `incr` with
332
+ -- no init returns nil for an absent key, which doubles as the existence probe,
333
+ -- so only a path unseen today reaches the cap check below.
334
+ --
335
+ -- Deliberately no init_ttl here: shdict rejects init_ttl unless init is also
336
+ -- given ("'init_ttl' must be used with 'init'"), and that raises a Lua error
337
+ -- which aborts the whole log handler — taking the status counters below down
338
+ -- with it. The TTL is set by the safe_add on the creating request instead.
339
+ if not analytics:incr(path_key, 1) then
340
+ local distinct = analytics:incr(gpfx .. "pn", 1, 0, D48H) or 0
341
+ if distinct > PATH_CARDINALITY_CAP then
342
+ analytics:incr(gpfx .. "p:other", 1, 0, D48H)
343
+ else
344
+ -- `pn` was already the distinct-path counter, so the slots just give it an
345
+ -- enumerable side. Stored WITHOUT the "p:" prefix; the reader adds it back.
346
+ if analytics:safe_add(path_key, 1, D48H) then
347
+ analytics:set(gpfx .. "pi:" .. distinct, norm, D48H)
348
+ end
349
+ end
350
+ end
351
+ end
352
+ end)
353
+
354
+ -- 3d. Status-code mix. Bounded by construction (a few dozen codes).
355
+ pcall(function()
356
+ bump_indexed(gpfx .. "s:" .. status, D48H, gpfx .. "sn", gpfx .. "si:", status)
357
+ end)
153
358
 
154
359
  -- ── 4. Raw request ring buffer ──────────────────────────────────────────────
155
360
 
156
361
  pcall(function()
157
362
  local ok_j, j = pcall(cjson.encode, {
158
363
  ip = ip,
364
+ -- nil is simply omitted by cjson, so a box with no GeoIP writes the same shape
365
+ -- minus this key rather than a null the reader has to special-case.
366
+ country = cc,
159
367
  ts = ts,
160
368
  method = method,
161
369
  status = status,
@@ -178,5 +386,5 @@ end)
178
386
  if pipe_log and pipe_log.pipe_request_log
179
387
  and request_data:get("log_pipe:sub:" .. host) then
180
388
  ngx.timer.at(0, pipe_log.pipe_request_log,
181
- host, ip, ts, ua, uri, req_len, bytes, rt, method, status)
389
+ host, ip, ts, ua, uri, req_len, bytes, rt, method, status, cc)
182
390
  end
@@ -0,0 +1,39 @@
1
+ -- Proof that a routing TARGET is ours, as Openship Cloud's shared edge now requires
2
+ -- before forwarding `<slug>.opsh.io` to it.
3
+ --
4
+ -- Keyed on (organization_id, target) because that is what the upstream keys it on:
5
+ -- one target serves every free domain on that box, so this is target-level state,
6
+ -- not domain-level. One install holds several rows — its own address plus one per
7
+ -- remote server it deploys to.
8
+ --
9
+ -- The token is persisted HERE, not only on the edge, because the upstream's list
10
+ -- endpoint returns id/target/status/expiry but NOT the token. Verification lasts 90
11
+ -- days and the upstream re-probes THE SAME token within 7 days of expiry; a 404 then
12
+ -- expires it and the free domain stops resolving ~83 days after a green deploy with
13
+ -- nothing in our logs. So a token that lived only on a box's disk would be
14
+ -- unrecoverable after a rebuild, and rows here are never deleted when a free domain
15
+ -- is dropped.
16
+ --
17
+ -- `server_id` is ON DELETE SET NULL, not cascade: losing the server row must not
18
+ -- take the token with it.
19
+ CREATE TABLE IF NOT EXISTS "edge_target_verification" (
20
+ "id" text PRIMARY KEY NOT NULL,
21
+ "organization_id" text NOT NULL REFERENCES "organization"("id") ON DELETE CASCADE,
22
+ "target" text NOT NULL,
23
+ "host" text NOT NULL,
24
+ "server_id" text REFERENCES "servers"("id") ON DELETE SET NULL,
25
+ "verification_id" integer,
26
+ "token" text,
27
+ "retired_tokens" jsonb,
28
+ "challenge_path" text,
29
+ "status" text NOT NULL DEFAULT 'pending',
30
+ "validated_ip" text,
31
+ "expires_at" timestamp,
32
+ "last_checked_at" timestamp,
33
+ "last_error" text,
34
+ "created_at" timestamp NOT NULL DEFAULT now(),
35
+ "updated_at" timestamp NOT NULL DEFAULT now()
36
+ );
37
+ --> statement-breakpoint
38
+ CREATE UNIQUE INDEX IF NOT EXISTS "uq_edge_target_verification"
39
+ ON "edge_target_verification" ("organization_id", "target");
@@ -0,0 +1,84 @@
1
+ -- Continuous container health monitoring: the durable memory behind the alerts.
2
+ --
3
+ -- Openship knew a container had crashed at exactly ONE moment — the ~15s post-deploy
4
+ -- stabilization watch. After that nobody looked again, so an OOM-kill at 3am or a
5
+ -- Postgres that starts bouncing after a host reboot reached the operator only when a
6
+ -- user complained. The health watch closes that, but a poller that alerts on "the
7
+ -- container isn't running" gets muted within a day: a redeploy recreating containers,
8
+ -- an operator's `docker stop`, an unreachable host, and a crash loop all look exactly
9
+ -- like a failure at a point in time.
10
+ --
11
+ -- So the unit of alerting is an INCIDENT, not a state reading: opened once, escalated
12
+ -- only when it gets worse, resolved once with a downtime duration. This table is that
13
+ -- memory — it survives the control-plane restart that in-process state would not (a box
14
+ -- down for three days must not re-page on every API restart), and it doubles as the
15
+ -- project Health tab's history.
16
+ CREATE TABLE IF NOT EXISTS "service_incident" (
17
+ "id" text PRIMARY KEY NOT NULL,
18
+ "organization_id" text NOT NULL REFERENCES "organization"("id") ON DELETE CASCADE,
19
+ -- Null for SERVER-scoped incidents (an unreachable box gets ONE incident, not one
20
+ -- per service on it — that fan-out is the single loudest source of alert fatigue).
21
+ "project_id" text REFERENCES "project"("id") ON DELETE CASCADE,
22
+ -- The resolved service row, when there is one. SET NULL, not cascade: deleting a
23
+ -- service must not erase the record of the outage it had (`service_key` still
24
+ -- identifies it, and `service_name` is snapshotted below).
25
+ "service_id" text REFERENCES "service"("id") ON DELETE SET NULL,
26
+ -- Stable identity across a container RECREATE, which is the whole point: every
27
+ -- deploy mints a new container id, so keying incidents on container id would open a
28
+ -- fresh incident for the same workload after each redeploy. Service id when we
29
+ -- resolved one, else the container name (single-app deploys have no service row);
30
+ -- `server:<id>` for server-scoped rows.
31
+ "service_key" text NOT NULL,
32
+ -- Snapshot of the display name at incident time — the alert text must still read
33
+ -- correctly after a rename, and history rows must not silently change meaning.
34
+ "service_name" text NOT NULL,
35
+ "server_id" text REFERENCES "servers"("id") ON DELETE SET NULL,
36
+ "container_id" text,
37
+ -- down | crash_loop | unhealthy | server_unreachable. Ordered by severity in the
38
+ -- watcher: an incident escalates upward (and re-notifies), never downward.
39
+ "kind" text NOT NULL,
40
+ "status" text NOT NULL DEFAULT 'open', -- open | resolved
41
+ "reason" text,
42
+ "exit_code" integer,
43
+ "restart_count" integer NOT NULL DEFAULT 0,
44
+ "oom_killed" boolean NOT NULL DEFAULT false,
45
+ -- Consecutive ticks that agreed on this verdict. An incident is only opened (and
46
+ -- only notified) at >= 2, so a snapshot taken mid-restart or during a slow
47
+ -- healthcheck start never pages anyone.
48
+ "confirmations" integer NOT NULL DEFAULT 0,
49
+ -- Notification bookkeeping. Deliberately a count, not a flag: open, escalation and
50
+ -- resolve each notify exactly once, and nothing else ever does.
51
+ "notify_count" integer NOT NULL DEFAULT 0,
52
+ "notified_at" timestamp,
53
+ "log_excerpt" text,
54
+ "opened_at" timestamp NOT NULL DEFAULT now(),
55
+ "resolved_at" timestamp,
56
+ -- Last tick that actually OBSERVED this workload. Not advanced when the host is
57
+ -- unreachable — during a connectivity loss we know nothing, so nothing is claimed.
58
+ "last_seen_at" timestamp NOT NULL DEFAULT now(),
59
+ "created_at" timestamp NOT NULL DEFAULT now(),
60
+ "updated_at" timestamp NOT NULL DEFAULT now()
61
+ );
62
+ --> statement-breakpoint
63
+ -- One open incident per workload — the dedup primitive. As a partial unique index,
64
+ -- "we are already alerting about this" is a DB invariant rather than sweep
65
+ -- bookkeeping: two overlapping ticks (a manual Run now landing on a cron tick) cannot
66
+ -- both open one and double-notify.
67
+ CREATE UNIQUE INDEX IF NOT EXISTS "uq_service_incident_open"
68
+ ON "service_incident" ("project_id", "service_key")
69
+ WHERE "status" = 'open' AND "project_id" IS NOT NULL;
70
+ --> statement-breakpoint
71
+ -- Same invariant for server-scoped rows. Needs its own index because Postgres treats
72
+ -- NULLs as distinct in a unique index, so the one above would happily allow a second
73
+ -- open row for the same box.
74
+ CREATE UNIQUE INDEX IF NOT EXISTS "uq_service_incident_open_server"
75
+ ON "service_incident" ("server_id")
76
+ WHERE "status" = 'open' AND "project_id" IS NULL;
77
+ --> statement-breakpoint
78
+ CREATE INDEX IF NOT EXISTS "idx_service_incident_project"
79
+ ON "service_incident" ("project_id", "opened_at" DESC);
80
+ --> statement-breakpoint
81
+ -- Intent marker. `disableProject` stopped the container and recorded NOTHING, so
82
+ -- "the operator turned this off" was unknowable — the health watch would page about
83
+ -- every deliberately-disabled project, forever. Set on disable, cleared on enable.
84
+ ALTER TABLE "project" ADD COLUMN IF NOT EXISTS "disabled_at" timestamp;
@@ -0,0 +1,72 @@
1
+ -- Widen the daily analytics rollup: distinct visitors, top paths, status mix.
2
+ --
3
+ -- The visitor pipeline already persisted per-country counts, and the dashboard
4
+ -- already rendered cards for "unique IPs" and a top-paths table. Neither was real:
5
+ -- `top_paths` was hardcoded `[]` in the service layer, and "unique IPs" was
6
+ -- `unique_requests` — the count of non-static REQUESTS — so five page views from
7
+ -- one browser read as five visitors. These columns are what makes those two
8
+ -- numbers mean what the UI has been claiming.
9
+ --
10
+ -- All three land HERE, on the daily table, rather than on the per-minute one:
11
+ -- visitors and paths are the high-cardinality metrics, and holding them per-minute
12
+ -- would multiply the edge's shared-dict key count by ~1440 to feed a chart nobody
13
+ -- asked for — evicting, in the process, the request counters they annotate.
14
+ -- Countries already live here for the same reason, and the edge keeps all four
15
+ -- under one `g:{domain}:{day}:` key prefix so a reader gets the whole day in a
16
+ -- single dict scan.
17
+ --
18
+ -- Additive with defaults on purpose: `migrations-additive.test.ts` rejects a
19
+ -- NOT NULL column without a DEFAULT, because an older cross-version dump omits the
20
+ -- column entirely and Drizzle then emits DEFAULT for it — with no default that's a
21
+ -- NULL into NOT NULL on the newer receiver, i.e. a broken cloud/project transfer.
22
+
23
+ -- Distinct visitors for the day, deduplicated AT THE EDGE and stored as a count.
24
+ --
25
+ -- Privacy is a property of where the dedup happens: the edge hashes each address
26
+ -- with a salt that is generated on that box, rotates daily, and never leaves its
27
+ -- shared memory. Only the cardinality of that set is written here. No address and
28
+ -- no per-visitor row exists at any layer, which is what keeps the published
29
+ -- "we don't collect behavioural analytics on your end users" true while still
30
+ -- reporting a real visitor number.
31
+ --
32
+ -- Understates on a very busy domain: the edge's `visitors` zone LRU-evicts past
33
+ -- roughly 1M distinct/day. mgmt_api `GET /status` reports that zone's free space so
34
+ -- a reader can label the number approximate instead of presenting an eviction
35
+ -- artifact as a measurement.
36
+ ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "visitors" integer DEFAULT 0 NOT NULL;
37
+ --> statement-breakpoint
38
+
39
+ -- Top paths: { "/": 900, "/orders/:id": 120, "other": 40 }.
40
+ --
41
+ -- Normalized at the edge before it ever becomes a key: query string stripped (so a
42
+ -- ?token= or ?session= can never be persisted here), numeric and UUID segments
43
+ -- collapsed to `:id`, key length capped, and the tail past a cardinality cap folded
44
+ -- into "other" — otherwise a scanner walking /wp-admin variants mints thousands of
45
+ -- permanent keys and evicts the real counters.
46
+ ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "paths" jsonb;
47
+ --> statement-breakpoint
48
+
49
+ -- Status-code mix: { "200": 4210, "404": 17, "502": 3 }. Status was captured into
50
+ -- the edge's raw-request ring buffer (RAM, 1h) but never aggregated, so error rate
51
+ -- was unanswerable from any persisted data. Bounded by construction.
52
+ ALTER TABLE "server_analytics_geo" ADD COLUMN IF NOT EXISTS "statuses" jsonb;
53
+ --> statement-breakpoint
54
+
55
+ -- Widen the per-minute bandwidth counters from int4 to int8.
56
+ --
57
+ -- These are bytes for ONE MINUTE, and int4 caps at 2,147,483,647 — about 2.1 GB,
58
+ -- which a single minute reaches at roughly 286 Mbps. That is unremarkable for a
59
+ -- site serving video or large downloads. Past it Postgres raises "integer out of
60
+ -- range", the upsert for that minute fails, and the scrape for the whole domain
61
+ -- dies with it — so the busiest domains on a box would be precisely the ones with
62
+ -- no analytics, and it would look like the feature simply didn't work for them.
63
+ --
64
+ -- Now that collection runs unattended on a schedule rather than only when someone
65
+ -- opens the tab, that failure would happen in a job with nobody reading the error.
66
+ --
67
+ -- Widening is lossless and there is no application change: the edge already
68
+ -- accumulates these as Lua numbers (doubles, exact to 2^53) and Drizzle reads them
69
+ -- back with mode:"number", which is exact over the same range.
70
+ ALTER TABLE "server_analytics" ALTER COLUMN "bandwidth_in" TYPE bigint;
71
+ --> statement-breakpoint
72
+ ALTER TABLE "server_analytics" ALTER COLUMN "bandwidth_out" TYPE bigint;
@@ -0,0 +1,68 @@
1
+ -- Resource usage history: the persisted series behind the Monitoring tab's
2
+ -- CPU/memory chart.
3
+ --
4
+ -- Usage existed only as a live 5-second SSE stream, so closing the tab discarded
5
+ -- everything. "Was memory climbing before the OOM at 3am" was unanswerable, which is
6
+ -- the question the data exists to answer.
7
+ --
8
+ -- ── Why this is a SAMPLED series, not a counter series like server_analytics ──
9
+ --
10
+ -- Traffic is counted for free: the edge increments a shared-dict key on every
11
+ -- request, so per-minute granularity costs nothing and is exact. Resource usage has
12
+ -- to be PROBED, and `docker stats` occupies the daemon for roughly a second per
13
+ -- container because it must collect two CPU samples to compute a delta — about 500x
14
+ -- a `docker inspect`. The sampling cadence therefore IS the resolution, and the cost
15
+ -- scales with container count. 5-minute buckets are the compromise between catching
16
+ -- a spike and not pinning every daemon on the estate; ~288 rows/day/service.
17
+ --
18
+ -- ── serviceKey is NOT NULL text, not a nullable service_id FK ────────────────
19
+ --
20
+ -- Postgres treats NULLs as DISTINCT in a unique index, so a nullable column cannot
21
+ -- carry the "this row is the project itself" case without the partial-index
22
+ -- workaround migration 0083 needed for service_incident. A sentinel ('__app__',
23
+ -- matching the health watch's own constant so both features name the same workload
24
+ -- identically) keeps one plain unique index correct.
25
+ --
26
+ -- Accepted consequence: deleting a service leaves its samples behind until retention
27
+ -- prunes them. That is the right trade — history should outlive the thing it
28
+ -- describes — and the project cascade still bounds the growth.
29
+ --
30
+ -- ── What is deliberately absent ─────────────────────────────────────────────
31
+ --
32
+ -- No project-total row. For a compose project the total is the per-bucket SUM of its
33
+ -- services, computed at read. Storing it too would duplicate rows that are free to
34
+ -- drift from their own sources, and would be wrong the moment a service is added or
35
+ -- removed mid-window.
36
+ CREATE TABLE IF NOT EXISTS "resource_usage" (
37
+ "id" text PRIMARY KEY NOT NULL,
38
+ "project_id" text NOT NULL REFERENCES "project"("id") ON DELETE CASCADE,
39
+ -- service.id, or '__app__' for a single-container project.
40
+ "service_key" text NOT NULL,
41
+ -- Bucket start in epoch MINUTES (not a timestamp), floored to a multiple of 5 —
42
+ -- same integer-bucket convention as server_analytics.
43
+ "minute" integer NOT NULL,
44
+ -- Per-CORE percent, as the runtimes report it: >100 is correct on a multi-core box
45
+ -- and must not be clamped. Readers divide by core count for "share of this host".
46
+ "cpu_percent" real DEFAULT 0 NOT NULL,
47
+ "memory_mb" real DEFAULT 0 NOT NULL,
48
+ -- CUMULATIVE since container start, not per-bucket deltas — that is what the
49
+ -- runtimes expose. Readers difference consecutive buckets and clamp negatives to
50
+ -- zero, because a restart resets the counter and would otherwise plot as a large
51
+ -- negative spike. bigint: a long-lived busy container passes int4 in days.
52
+ -- Always 0 for bare deploys (per-process accounting needs eBPF or a netns), so a
53
+ -- reader must not present 0 here as "no network traffic".
54
+ "network_rx_bytes" bigint DEFAULT 0 NOT NULL,
55
+ "network_tx_bytes" bigint DEFAULT 0 NOT NULL,
56
+ "created_at" timestamp DEFAULT now() NOT NULL
57
+ );
58
+ --> statement-breakpoint
59
+
60
+ -- Makes a repeated sample for the same bucket a no-op (insert ... on conflict do
61
+ -- nothing) rather than a double count, so the whole sweep is safely re-runnable.
62
+ CREATE UNIQUE INDEX IF NOT EXISTS "uq_resource_usage_project_service_minute"
63
+ ON "resource_usage" ("project_id", "service_key", "minute");
64
+ --> statement-breakpoint
65
+
66
+ -- Every read is "this project over this window".
67
+ CREATE INDEX IF NOT EXISTS "idx_resource_usage_project_minute"
68
+ ON "resource_usage" ("project_id", "minute");
@@ -0,0 +1,41 @@
1
+ -- Remote infra (edge / mail container) update tracking.
2
+ --
3
+ -- When the control plane's APP_VERSION moves forward, the edge/mail containers
4
+ -- already deployed on remote servers stay on their old image tag. Two additions
5
+ -- drive the "monitor → advise → apply" loop for that drift:
6
+ --
7
+ -- instance_settings.auto_update_infra / last_seen_version — the instance-wide
8
+ -- auto-update toggle and the version the infra reconcile last ran for (so the
9
+ -- boot hook fires exactly once per upgrade). Both additive: the boolean
10
+ -- carries a DEFAULT, last_seen_version is nullable.
11
+ --
12
+ -- server_container_status — the container sibling of server_module_status.
13
+ -- One row per (server, component); "behind" is a plain tag comparison
14
+ -- (running image ref != pinned image ref), since infra images are pinned to
15
+ -- APP_VERSION. Mirrors the module table's org+behind index and unique key.
16
+ ALTER TABLE "instance_settings" ADD COLUMN IF NOT EXISTS "auto_update_infra" boolean DEFAULT false NOT NULL;
17
+ --> statement-breakpoint
18
+ ALTER TABLE "instance_settings" ADD COLUMN IF NOT EXISTS "last_seen_version" text;
19
+ --> statement-breakpoint
20
+ CREATE TABLE IF NOT EXISTS "server_container_status" (
21
+ "id" text PRIMARY KEY NOT NULL,
22
+ "organization_id" text REFERENCES "organization"("id") ON DELETE CASCADE,
23
+ "server_id" text NOT NULL REFERENCES "servers"("id") ON DELETE CASCADE,
24
+ "component" text NOT NULL,
25
+ "running_label" text,
26
+ "pinned_label" text,
27
+ "running_version" text,
28
+ "pinned_version" text,
29
+ "behind" boolean DEFAULT false NOT NULL,
30
+ "latest_in_progress" boolean DEFAULT false NOT NULL,
31
+ "detail" jsonb,
32
+ "checked_at" timestamp DEFAULT now() NOT NULL,
33
+ "created_at" timestamp DEFAULT now() NOT NULL,
34
+ "updated_at" timestamp DEFAULT now() NOT NULL
35
+ );
36
+ --> statement-breakpoint
37
+ CREATE UNIQUE INDEX IF NOT EXISTS "uq_server_container_status"
38
+ ON "server_container_status" ("server_id", "component");
39
+ --> statement-breakpoint
40
+ CREATE INDEX IF NOT EXISTS "idx_server_container_status_org_behind"
41
+ ON "server_container_status" ("organization_id", "behind");
@@ -0,0 +1,17 @@
1
+ -- Repair: add `server_container_status.running_version` / `pinned_version` on a
2
+ -- database that ran an earlier form of 0086.
3
+ --
4
+ -- 0086 was edited in place after it had already been applied: the version columns
5
+ -- were added to its CREATE TABLE, but that create is `IF NOT EXISTS`, so on any
6
+ -- database that had already run the older 0086 the table exists and the amended
7
+ -- create is a silent no-op — the columns never appear. drizzle's migrator then
8
+ -- refuses to re-run 0086 (its `when` equals the last-applied stamp, and the
9
+ -- comparison is strict `<`; see 0081_project_readiness_repair for the same trap),
10
+ -- so nothing else patches it either. Every `listByOrg` then dies with
11
+ -- `column "running_version" does not exist` (42703).
12
+ --
13
+ -- Idempotent and stamped after 0086, so it sorts as a fresh migration: a database
14
+ -- created cleanly gets the columns at 0086 and this is a no-op; a database that
15
+ -- ran the stale 0086 gets them here. Safe to keep in history permanently.
16
+ ALTER TABLE "server_container_status" ADD COLUMN IF NOT EXISTS "running_version" text;--> statement-breakpoint
17
+ ALTER TABLE "server_container_status" ADD COLUMN IF NOT EXISTS "pinned_version" text;
@@ -0,0 +1,27 @@
1
+ -- Audit log: call source + per-org recording switch.
2
+ --
3
+ -- audit_event.source — WHERE the action came in from: "dashboard", "mcp",
4
+ -- "cli", "api", "webhook", "system". Nullable on purpose: rows written
5
+ -- before this migration have no knowable source and must not be
6
+ -- mislabelled as any particular one, so they render as "Unknown". This is
7
+ -- what makes "show me only what the AI assistant did" answerable at all —
8
+ -- MCP tool calls re-enter the app through app.fetch() with nothing but an
9
+ -- Authorization header, so until now an MCP-driven write and a CLI write
10
+ -- produced identical rows.
11
+ --
12
+ -- audit_settings — recording on/off + retention, PER ORGANIZATION. Not
13
+ -- instance_settings (one row would let a single tenant disable auditing for
14
+ -- every org on a CLOUD_MODE instance) and not organization.metadata (Better
15
+ -- Auth owns those writes, which is why the auditRetentionDays documented
16
+ -- there never got a writer). Default enabled = true: auditing stays on
17
+ -- unless an admin turns it off.
18
+ ALTER TABLE "audit_event" ADD COLUMN IF NOT EXISTS "source" text;
19
+ --> statement-breakpoint
20
+ CREATE INDEX IF NOT EXISTS "audit_event_org_source_idx" ON "audit_event" ("organization_id","source","created_at" DESC);
21
+ --> statement-breakpoint
22
+ CREATE TABLE IF NOT EXISTS "audit_settings" (
23
+ "organization_id" text PRIMARY KEY NOT NULL REFERENCES "organization"("id") ON DELETE CASCADE,
24
+ "enabled" boolean DEFAULT true NOT NULL,
25
+ "retention_days" integer DEFAULT 90 NOT NULL,
26
+ "updated_at" timestamp DEFAULT now() NOT NULL
27
+ );