@andrian.yablonskyy/thub-coordinator 1.0.38 → 1.0.40
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/thub-admin.js +26 -5
- package/config.json +5 -6
- package/package.json +2 -2
- package/public/js/actions.js +58 -0
- package/public/js/log-viewer.js +188 -56
- package/public/js/settings.js +66 -0
- package/scripts/install-default-config.js +1 -1
- package/scripts/install-target.js +2 -2
- package/src/api/admin.js +1 -1
- package/src/api/agent.js +28 -22
- package/src/api/resource.js +11 -14
- package/src/api/sse.js +13 -10
- package/src/config.js +33 -10
- package/src/db/migrations/022_add_sessions.sql +10 -0
- package/src/db/migrations/023_drop_artifacts.sql +4 -0
- package/src/db/migrations/024_add_settings.sql +10 -0
- package/src/db/migrations/025_add_job_artifacts.sql +4 -0
- package/src/dev/virtual.js +10 -6
- package/src/login-guard.js +111 -0
- package/src/rate-limit.js +137 -0
- package/src/security.js +104 -0
- package/src/server.js +53 -23
- package/src/services/admin-users.js +32 -19
- package/src/services/cleanup.js +2 -2
- package/src/services/heartbeat.js +23 -3
- package/src/services/job-artifacts.js +91 -0
- package/src/services/jobs.js +17 -9
- package/src/services/logs.js +12 -14
- package/src/services/registry.js +34 -0
- package/src/services/retention.js +5 -14
- package/src/services/session-store.js +127 -0
- package/src/services/settings.js +345 -0
- package/src/web/routes.js +230 -21
- package/test/admin-users.test.js +16 -16
- package/test/agent-visibility.test.js +85 -0
- package/test/cleanup.test.js +1 -5
- package/test/groups-tab.test.js +97 -0
- package/test/job-artifacts.test.js +115 -0
- package/test/login-guard.test.js +139 -0
- package/test/logs.test.js +150 -0
- package/test/rate-limit.test.js +106 -0
- package/test/scheduler.test.js +10 -12
- package/test/security.test.js +107 -0
- package/test/session.test.js +132 -0
- package/test/settings.test.js +206 -0
- package/views/admin/settings.pug +133 -0
- package/views/groups/list.pug +1 -1
- package/views/help/_agent-cli.pug +2 -2
- package/views/help/_agent-setup.pug +4 -1
- package/views/help/_client-setup.pug +1 -1
- package/views/help/_coordinator.pug +18 -7
- package/views/help/_docker.pug +1 -1
- package/views/help/_env.pug +2 -1
- package/views/help/_git.pug +3 -2
- package/views/help/_overview.pug +2 -2
- package/views/help/_troubleshooting.pug +4 -0
- package/views/jobs/list.pug +5 -5
- package/views/jobs/show.pug +33 -7
- package/views/layout.pug +4 -1
- package/views/mixins/client-config.pug +65 -3
- package/views/mixins/list-controls.pug +1 -1
- package/views/mixins/log-viewer.pug +14 -0
- package/views/mixins/resource.pug +1 -8
- package/src/services/artifacts.js +0 -149
package/src/api/agent.js
CHANGED
|
@@ -24,7 +24,21 @@ function createAgentRouter({ services, config }){
|
|
|
24
24
|
// the /api/v1 prefix with the resource and admin routers — a blanket
|
|
25
25
|
// router-level middleware would run for their paths too, before route
|
|
26
26
|
// matching even happens, and reject them for lacking an agent token.
|
|
27
|
-
auth = requireRole('agent')
|
|
27
|
+
auth = requireRole('agent'),
|
|
28
|
+
|
|
29
|
+
// Which jobs an agent may see (README §12): a `ci` token (pipelines) all
|
|
30
|
+
// of them; a `cli` token (a person's own `thub`) only the jobs it
|
|
31
|
+
// submitted. Anyone else's job is answered exactly like a job that
|
|
32
|
+
// doesn't exist (404), so a cli token can't tell which ids are in use.
|
|
33
|
+
canSee = (agent, job) => agent.kind === 'ci' || job.agent_id === agent.id,
|
|
34
|
+
visibleJob = (req, res, next) => {
|
|
35
|
+
const job = services.jobs.get(req.params.id);
|
|
36
|
+
if (!job || !canSee(req.agent, job)){
|
|
37
|
+
return res.status(404).json({ error: 'Unknown job' });
|
|
38
|
+
}
|
|
39
|
+
req.job = job;
|
|
40
|
+
next();
|
|
41
|
+
};
|
|
28
42
|
|
|
29
43
|
router.post('/jobs', auth, (req, res, next) => {
|
|
30
44
|
try {
|
|
@@ -40,12 +54,9 @@ function createAgentRouter({ services, config }){
|
|
|
40
54
|
}
|
|
41
55
|
});
|
|
42
56
|
|
|
43
|
-
router.get('/jobs/:id', auth, (req, res) => {
|
|
44
|
-
const job =
|
|
45
|
-
|
|
46
|
-
return res.status(404).json({ error: 'Unknown job' });
|
|
47
|
-
}
|
|
48
|
-
const resource = job.resource_id ? services.registry.get(job.resource_id) : null;
|
|
57
|
+
router.get('/jobs/:id', auth, visibleJob, (req, res) => {
|
|
58
|
+
const { job } = req,
|
|
59
|
+
resource = job.resource_id ? services.registry.get(job.resource_id) : null;
|
|
49
60
|
res.json({
|
|
50
61
|
...services.jobs.publicJob(job),
|
|
51
62
|
resource: resource ? { id: resource.id, name: resource.name } : job.resource_name ? { id: null, name: job.resource_name } : null
|
|
@@ -57,13 +68,14 @@ function createAgentRouter({ services, config }){
|
|
|
57
68
|
state: req.query.state,
|
|
58
69
|
source: req.query.source,
|
|
59
70
|
agentId: req.agent.id,
|
|
60
|
-
|
|
71
|
+
// A cli token's list is always its own jobs (canSee above).
|
|
72
|
+
mine: req.agent.kind !== 'ci' || req.query.mine === 'true' || req.query.mine === '1',
|
|
61
73
|
limit: req.query.limit ? Number(req.query.limit) : undefined
|
|
62
74
|
});
|
|
63
75
|
res.json({ jobs: jobs.map(services.jobs.publicJob) });
|
|
64
76
|
});
|
|
65
77
|
|
|
66
|
-
router.post('/jobs/:id/cancel', auth, (req, res, next) => {
|
|
78
|
+
router.post('/jobs/:id/cancel', auth, visibleJob, (req, res, next) => {
|
|
67
79
|
try {
|
|
68
80
|
const job = services.jobs.cancel(req.params.id, { agentId: req.agent.id, isAdmin: false });
|
|
69
81
|
res.json(services.jobs.publicJob(job));
|
|
@@ -73,26 +85,20 @@ function createAgentRouter({ services, config }){
|
|
|
73
85
|
}
|
|
74
86
|
});
|
|
75
87
|
|
|
76
|
-
router.get('/jobs/:id/logs', auth, (req, res) => {
|
|
88
|
+
router.get('/jobs/:id/logs', auth, visibleJob, (req, res) => {
|
|
77
89
|
const after = Number(req.query.after || 0),
|
|
78
90
|
limit = req.query.limit ? Number(req.query.limit) : undefined;
|
|
79
91
|
res.json({ lines: services.logs.listSince(req.params.id, after, limit) });
|
|
80
92
|
});
|
|
81
93
|
|
|
82
|
-
router.get('/jobs/:id/logs/stream', auth, (req, res) => {
|
|
83
|
-
attachJobStream(req, res, { jobId: req.params.id, services
|
|
94
|
+
router.get('/jobs/:id/logs/stream', auth, visibleJob, (req, res) => {
|
|
95
|
+
attachJobStream(req, res, { jobId: req.params.id, services });
|
|
84
96
|
});
|
|
85
97
|
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
size: a.size,
|
|
91
|
-
sha256: a.sha256,
|
|
92
|
-
contentType: a.content_type,
|
|
93
|
-
url: services.artifacts.signedUrl(a)
|
|
94
|
-
}));
|
|
95
|
-
res.json({ artifacts });
|
|
98
|
+
// What the job reported (README §7.3): metadata and links only. `url` is
|
|
99
|
+
// `link` again, for Agents from before, which print `url`.
|
|
100
|
+
router.get('/jobs/:id/artifacts', auth, visibleJob, (req, res) => {
|
|
101
|
+
res.json({ artifacts: req.job.artifacts.map((a) => ({ ...a, url: a.link })) });
|
|
96
102
|
});
|
|
97
103
|
|
|
98
104
|
router.get('/resources', auth, (req, res) => {
|
package/src/api/resource.js
CHANGED
|
@@ -14,12 +14,9 @@
|
|
|
14
14
|
'use strict';
|
|
15
15
|
|
|
16
16
|
const express = require('express'),
|
|
17
|
-
multer = require('multer'),
|
|
18
17
|
{ requireRole, requireJoinKey, appVersion } = require('../auth'),
|
|
19
18
|
{ JOB_STATES } = require('@andrian.yablonskyy/thub-common');
|
|
20
19
|
|
|
21
|
-
const upload = multer({ dest: require('node:os').tmpdir() });
|
|
22
|
-
|
|
23
20
|
function requireOwnResource(req, res, next){
|
|
24
21
|
if (req.resource.id !== req.params.id){
|
|
25
22
|
return res.status(403).json({ error: 'Token does not match resource' });
|
|
@@ -130,7 +127,9 @@ function createResourceRouter({ services, config }){
|
|
|
130
127
|
if (req.resource.remove_requested_at && !activeJobId){
|
|
131
128
|
services.jobs.completeRemoval(req.params.id, { by: 'removal-confirmed' });
|
|
132
129
|
}
|
|
133
|
-
|
|
130
|
+
// The current interval in every reply, not only at registration, so a
|
|
131
|
+
// change from the dashboard (§13.2) reaches running Clients at once.
|
|
132
|
+
res.json({ serverTime: new Date().toISOString(), commands, heartbeatIntervalSec: config.heartbeat.intervalSec });
|
|
134
133
|
}
|
|
135
134
|
catch (err){
|
|
136
135
|
next(err);
|
|
@@ -211,20 +210,18 @@ function createResourceRouter({ services, config }){
|
|
|
211
210
|
}
|
|
212
211
|
});
|
|
213
212
|
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
next(err);
|
|
221
|
-
}
|
|
213
|
+
// Artifacts are no longer stored on the Coordinator (README §9). A Client
|
|
214
|
+
// older than that still uploads its results after each job and would end
|
|
215
|
+
// the job as ERROR on a 404, so accept the upload and discard it unread.
|
|
216
|
+
router.post('/jobs/:id/artifacts', auth, requireOwnJob(services), (req, res) => {
|
|
217
|
+
req.resume();
|
|
218
|
+
req.on('end', () => res.status(201).json({ artifacts: [] }));
|
|
222
219
|
});
|
|
223
220
|
|
|
224
221
|
router.post('/jobs/:id/result', auth, requireOwnJob(services), (req, res, next) => {
|
|
225
222
|
try {
|
|
226
|
-
const { state, exitCode, summary } = req.body,
|
|
227
|
-
job = services.jobs.applyResult(req.params.id, req.resource.id, { state, exitCode, summary });
|
|
223
|
+
const { state, exitCode, summary, artifacts } = req.body,
|
|
224
|
+
job = services.jobs.applyResult(req.params.id, req.resource.id, { state, exitCode, summary, artifacts });
|
|
228
225
|
res.json(job);
|
|
229
226
|
}
|
|
230
227
|
catch (err){
|
package/src/api/sse.js
CHANGED
|
@@ -15,11 +15,12 @@
|
|
|
15
15
|
|
|
16
16
|
const { TERMINAL_JOB_STATES, JOB_STATES } = require('@andrian.yablonskyy/thub-common');
|
|
17
17
|
|
|
18
|
-
const WAITING_CHECK_MS = 15_000
|
|
18
|
+
const WAITING_CHECK_MS = 15_000,
|
|
19
|
+
REPLAY_PAGE = 2000;
|
|
19
20
|
|
|
20
21
|
// §6.4: log / state / end events, resumable via Last-Event-ID (§7: "the
|
|
21
22
|
// Agent reconnects with exponential backoff and resumes from the last seq").
|
|
22
|
-
function attachJobStream(req, res, { jobId, services
|
|
23
|
+
function attachJobStream(req, res, { jobId, services }){
|
|
23
24
|
const { jobs, logs, registry, bus } = services,
|
|
24
25
|
|
|
25
26
|
job = jobs.get(jobId);
|
|
@@ -47,17 +48,19 @@ function attachJobStream(req, res, { jobId, services, config }){
|
|
|
47
48
|
res.write(`data: ${JSON.stringify(data)}\n\n`);
|
|
48
49
|
}
|
|
49
50
|
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
51
|
+
// Replay everything after `afterSeq`, in pages (listSince stops at 500 a
|
|
52
|
+
// call — replaying one page used to skip lines 501.. of a longer log). All
|
|
53
|
+
// synchronous, so no live line can slip in between replay and subscribing.
|
|
54
|
+
for (let seq = afterSeq, page; (page = logs.listSince(jobId, seq, REPLAY_PAGE)).length;){
|
|
55
|
+
for (const line of page){
|
|
56
|
+
writeEvent('log', { ts: line.ts, stream: line.stream, line: line.line }, line.seq);
|
|
57
|
+
}
|
|
58
|
+
seq = page.at(-1).seq;
|
|
56
59
|
}
|
|
57
60
|
|
|
58
61
|
const current = jobs.get(jobId);
|
|
59
62
|
if (TERMINAL_JOB_STATES.has(current.state)){
|
|
60
|
-
writeEvent('end', { state: current.state
|
|
63
|
+
writeEvent('end', { state: current.state });
|
|
61
64
|
return res.end();
|
|
62
65
|
}
|
|
63
66
|
|
|
@@ -97,7 +100,7 @@ function attachJobStream(req, res, { jobId, services, config }){
|
|
|
97
100
|
if (evt.jobId !== jobId){
|
|
98
101
|
return;
|
|
99
102
|
}
|
|
100
|
-
writeEvent('end', { state: evt.state
|
|
103
|
+
writeEvent('end', { state: evt.state });
|
|
101
104
|
cleanup();
|
|
102
105
|
res.end();
|
|
103
106
|
};
|
package/src/config.js
CHANGED
|
@@ -28,6 +28,12 @@ const DEFAULTS = {
|
|
|
28
28
|
trustProxy: 'loopback',
|
|
29
29
|
dataDir: path.join(process.cwd(), '.data'),
|
|
30
30
|
sessionSecret: 'dev-only-change-me',
|
|
31
|
+
session: {
|
|
32
|
+
// The dashboard session cookie's Secure flag (server.js cookieSecure):
|
|
33
|
+
// "auto" = Secure whenever publicUrl is https:// (else per request:
|
|
34
|
+
// Secure only on a connection that is itself HTTPS); true / false force it.
|
|
35
|
+
secureCookie: 'auto'
|
|
36
|
+
},
|
|
31
37
|
// Shared secret Clients present to self-register (see api/resource.js).
|
|
32
38
|
// null disables auto-registration entirely — set it explicitly to turn it on.
|
|
33
39
|
clientJoinKey: null,
|
|
@@ -46,19 +52,23 @@ const DEFAULTS = {
|
|
|
46
52
|
defaultTimeoutSec: 1800,
|
|
47
53
|
maxTimeoutSec: 14400
|
|
48
54
|
},
|
|
55
|
+
// Requests per minute per caller (rate-limit.js, README §12): an agent or
|
|
56
|
+
// resource token or a dashboard user, else the client IP; sign-in attempts
|
|
57
|
+
// per IP on top. 0 turns a limit off.
|
|
58
|
+
rateLimit: {
|
|
59
|
+
requestsPerMinute: 500,
|
|
60
|
+
loginPerMinute: 10
|
|
61
|
+
},
|
|
62
|
+
// logRetentionDays, artifactRetentionDays and the `artifacts` section of
|
|
63
|
+
// older configs are no longer read: a job's log lines live as long as the
|
|
64
|
+
// job, and the Coordinator stores no artifacts (README §9).
|
|
49
65
|
retention: {
|
|
50
|
-
|
|
51
|
-
artifactRetentionDays: 30,
|
|
52
|
-
// How long a finished job (with its logs, artifacts and events) is kept
|
|
66
|
+
// How long a finished job (with its logs and events) is kept
|
|
53
67
|
// before the hourly retention task wipes it (services/cleanup.js):
|
|
54
68
|
// 1w | 2w | 1m | 3m | 6m | forever. `forever` keeps everything until a
|
|
55
69
|
// manual cleanup (dashboard Clean up database / thub-admin jobs clean).
|
|
56
70
|
jobRetention: 'forever'
|
|
57
71
|
},
|
|
58
|
-
artifacts: {
|
|
59
|
-
maxUploadMb: 512,
|
|
60
|
-
linkTtlHours: 168
|
|
61
|
-
},
|
|
62
72
|
// New-version check of the Coordinator/Agent/Client packages (README
|
|
63
73
|
// §10.2), every 15 minutes so a new Coordinator's "Update app" button
|
|
64
74
|
// shows up promptly. 0 disables the periodic check; registry null = the
|
|
@@ -103,7 +113,9 @@ function loadConfig(configPath = process.env.THUB_COORDINATOR_CONFIG){
|
|
|
103
113
|
(p) => p && fs.existsSync(p)
|
|
104
114
|
),
|
|
105
115
|
fileConfig = candidate ? JSON.parse(fs.readFileSync(candidate, 'utf8')) || {} : {},
|
|
106
|
-
config
|
|
116
|
+
// A copy: the dashboard's settings change the loaded config in place
|
|
117
|
+
// (services/settings.js), which must never reach DEFAULTS itself.
|
|
118
|
+
config = deepMerge(structuredClone(DEFAULTS), fileConfig);
|
|
107
119
|
|
|
108
120
|
// Environment overrides for the bits you don't want in a committed file.
|
|
109
121
|
if (process.env.THUB_LISTEN){
|
|
@@ -122,6 +134,19 @@ function loadConfig(configPath = process.env.THUB_COORDINATOR_CONFIG){
|
|
|
122
134
|
config.clientJoinKey = process.env.THUB_CLIENT_JOIN_KEY;
|
|
123
135
|
}
|
|
124
136
|
|
|
137
|
+
for (const key of ['requestsPerMinute', 'loginPerMinute']){
|
|
138
|
+
const v = config.rateLimit[key];
|
|
139
|
+
if (!Number.isInteger(v) || v < 0){
|
|
140
|
+
throw new Error(`rateLimit.${key} must be a whole number, 0 or more (0 = no limit; got ${JSON.stringify(v)}) in ${candidate || 'the config'}`);
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
if (![true, false, 'auto'].includes(config.session.secureCookie)){
|
|
145
|
+
throw new Error(
|
|
146
|
+
`session.secureCookie must be "auto", true or false (got ${JSON.stringify(config.session.secureCookie)}) in ${candidate || 'the config'}`
|
|
147
|
+
);
|
|
148
|
+
}
|
|
149
|
+
|
|
125
150
|
if (!Object.hasOwn(JOB_RETENTION, config.retention.jobRetention)){
|
|
126
151
|
throw new Error(
|
|
127
152
|
`retention.jobRetention must be one of: ${Object.keys(JOB_RETENTION).join(', ')} ` +
|
|
@@ -136,12 +161,10 @@ function loadConfig(configPath = process.env.THUB_COORDINATOR_CONFIG){
|
|
|
136
161
|
config.host = host;
|
|
137
162
|
config.port = Number(port);
|
|
138
163
|
config.dbPath = path.join(config.dataDir, 'thub.db');
|
|
139
|
-
config.artifactsDir = path.join(config.dataDir, 'artifacts');
|
|
140
164
|
config.workDir = path.join(config.dataDir, 'work');
|
|
141
165
|
config.avatarsDir = path.join(config.dataDir, 'avatars');
|
|
142
166
|
|
|
143
167
|
ensureWritableDir(config.dataDir, config.configPath);
|
|
144
|
-
ensureWritableDir(config.artifactsDir, config.configPath);
|
|
145
168
|
ensureWritableDir(config.avatarsDir, config.configPath);
|
|
146
169
|
|
|
147
170
|
return config;
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
-- Dashboard sessions (services/session-store.js), instead of express-session's
|
|
2
|
+
-- in-memory store: they survive a restart, and expired ones are pruned.
|
|
3
|
+
-- `expires` is epoch milliseconds (the session cookie's own expiry).
|
|
4
|
+
CREATE TABLE sessions (
|
|
5
|
+
sid TEXT PRIMARY KEY,
|
|
6
|
+
sess TEXT NOT NULL,
|
|
7
|
+
expires INTEGER NOT NULL
|
|
8
|
+
);
|
|
9
|
+
|
|
10
|
+
CREATE INDEX idx_sessions_expires ON sessions(expires);
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
-- Coordinator settings changed from the dashboard (services/settings.js,
|
|
2
|
+
-- /admin/settings). They override the config file — which the sandboxed
|
|
3
|
+
-- service can't write — and are overridden by environment variables.
|
|
4
|
+
-- `value` is JSON.
|
|
5
|
+
CREATE TABLE settings (
|
|
6
|
+
key TEXT PRIMARY KEY,
|
|
7
|
+
value TEXT NOT NULL,
|
|
8
|
+
updated_at TEXT NOT NULL,
|
|
9
|
+
updated_by TEXT
|
|
10
|
+
);
|
package/src/dev/virtual.js
CHANGED
|
@@ -170,10 +170,9 @@ function seedHistory(services, agentId, resourceIds){
|
|
|
170
170
|
// stay IDLE instead of going OUT_OF_SERVICE) and play out any job the
|
|
171
171
|
// scheduler assigns them — accept, PREPARING, RUNNING with streamed log
|
|
172
172
|
// lines, then a PASSED (mostly) or FAILED result — through the same
|
|
173
|
-
// services the Client API uses, so SSE
|
|
173
|
+
// services the Client API uses, so SSE and durations all work.
|
|
174
174
|
function startVirtualClients(services, config){
|
|
175
175
|
const { registry, jobs, logs, bus } = services,
|
|
176
|
-
intervalMs = config.heartbeat.intervalSec * 1000,
|
|
177
176
|
clients = new Map(); // resourceId -> { def, activity, running }
|
|
178
177
|
|
|
179
178
|
for (const def of VIRTUAL_CLIENTS){
|
|
@@ -232,9 +231,14 @@ function startVirtualClients(services, config){
|
|
|
232
231
|
}
|
|
233
232
|
}
|
|
234
233
|
},
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
234
|
+
// Like a real Client, follow heartbeat.intervalSec as it changes (§13.2).
|
|
235
|
+
tick = () => {
|
|
236
|
+
beat();
|
|
237
|
+
timer = setTimeout(tick, config.heartbeat.intervalSec * 1000);
|
|
238
|
+
timer.unref?.();
|
|
239
|
+
};
|
|
240
|
+
let timer = null;
|
|
241
|
+
tick();
|
|
238
242
|
|
|
239
243
|
const sleep = (ms) => new Promise((r) => setTimeout(r, ms)),
|
|
240
244
|
stillActive = (jobId) => ACTIVE_JOB_STATES.has(jobs.get(jobId)?.state);
|
|
@@ -287,7 +291,7 @@ function startVirtualClients(services, config){
|
|
|
287
291
|
|
|
288
292
|
return {
|
|
289
293
|
resourceIds: Object.fromEntries([...clients].map(([id, c]) => [c.def.type, id])),
|
|
290
|
-
stop: () =>
|
|
294
|
+
stop: () => clearTimeout(timer)
|
|
291
295
|
};
|
|
292
296
|
}
|
|
293
297
|
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file packages/coordinator/src/login-guard.js
|
|
3
|
+
* @description Sign-in protection: exponential backoff after failed attempts, per IP and per username, and a cap on concurrent password checks
|
|
4
|
+
*
|
|
5
|
+
* @author Andrian Yablonskyy
|
|
6
|
+
* @copyright Copyright (c) 2026 Andrian Yablonskyy. All rights reserved.
|
|
7
|
+
*
|
|
8
|
+
* This file is part of TestHub and is proprietary and confidential.
|
|
9
|
+
* Unauthorized copying, modification, distribution, or use of this file,
|
|
10
|
+
* via any medium, is strictly prohibited without prior written permission
|
|
11
|
+
* from AdSystem.PRO.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
'use strict';
|
|
15
|
+
|
|
16
|
+
// README §12. On top of the per-IP budget for sign-in attempts
|
|
17
|
+
// (rate-limit.js): after a few failures in a row, each further failure
|
|
18
|
+
// doubles the wait before the next attempt is even considered — checked
|
|
19
|
+
// before any password hashing, so a locked-out guesser costs nothing.
|
|
20
|
+
//
|
|
21
|
+
// - per IP: from the 5th failure, 1 s, 2 s, 4 s … up to 15 min.
|
|
22
|
+
// - per username: from the 10th, up to 5 min — milder, since anyone can
|
|
23
|
+
// trigger it for a name they know; it still slows guessing one account
|
|
24
|
+
// from many addresses to a crawl, without locking its owner out for long.
|
|
25
|
+
// Applied to any name typed, existing or not, so it reveals nothing.
|
|
26
|
+
//
|
|
27
|
+
// A success clears both. A record without a failure for an hour is dropped.
|
|
28
|
+
const POLICIES = {
|
|
29
|
+
ip: { after: 5, maxSec: 15 * 60 },
|
|
30
|
+
user: { after: 10, maxSec: 5 * 60 }
|
|
31
|
+
},
|
|
32
|
+
FORGET_MS = 60 * 60 * 1000,
|
|
33
|
+
MAX_RECORDS = 50_000,
|
|
34
|
+
// Password checks running at once (scrypt on libuv's 4-thread pool, which
|
|
35
|
+
// file serving shares): beyond this, "busy, retry" instead of queueing.
|
|
36
|
+
MAX_CONCURRENT = 8;
|
|
37
|
+
|
|
38
|
+
function createLoginGuard({ now = Date.now, maxConcurrent = MAX_CONCURRENT } = {}){
|
|
39
|
+
const records = new Map();
|
|
40
|
+
let inFlight = 0;
|
|
41
|
+
|
|
42
|
+
const keysOf = (ip, username) => [
|
|
43
|
+
['ip', `ip:${ip}`],
|
|
44
|
+
['user', `user:${String(username ?? '').trim().toLowerCase().slice(0, 128)}`]
|
|
45
|
+
];
|
|
46
|
+
|
|
47
|
+
function record(key){
|
|
48
|
+
const r = records.get(key);
|
|
49
|
+
if (r && now() - r.lastFailure > FORGET_MS){
|
|
50
|
+
records.delete(key);
|
|
51
|
+
return null;
|
|
52
|
+
}
|
|
53
|
+
return r || null;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
// Before checking a password: { ok } or { ok: false, retryAfterSec, reason }.
|
|
57
|
+
function check(ip, username){
|
|
58
|
+
let wait = 0;
|
|
59
|
+
for (const [, key]of keysOf(ip, username)){
|
|
60
|
+
const r = record(key);
|
|
61
|
+
if (r && r.lockedUntil > now()){
|
|
62
|
+
wait = Math.max(wait, r.lockedUntil - now());
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
if (wait){
|
|
66
|
+
return { ok: false, reason: 'backoff', retryAfterSec: Math.ceil(wait / 1000) };
|
|
67
|
+
}
|
|
68
|
+
if (inFlight >= maxConcurrent){
|
|
69
|
+
return { ok: false, reason: 'busy', retryAfterSec: 1 };
|
|
70
|
+
}
|
|
71
|
+
return { ok: true };
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// Wraps the password check itself, for the concurrency cap.
|
|
75
|
+
async function run(fn){
|
|
76
|
+
inFlight += 1;
|
|
77
|
+
try {
|
|
78
|
+
return await fn();
|
|
79
|
+
}
|
|
80
|
+
finally {
|
|
81
|
+
inFlight -= 1;
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
function failure(ip, username){
|
|
86
|
+
for (const [kind, key]of keysOf(ip, username)){
|
|
87
|
+
const { after, maxSec } = POLICIES[kind],
|
|
88
|
+
r = record(key) || { failures: 0, lockedUntil: 0, lastFailure: 0 };
|
|
89
|
+
r.failures += 1;
|
|
90
|
+
r.lastFailure = now();
|
|
91
|
+
if (r.failures >= after){
|
|
92
|
+
r.lockedUntil = now() + Math.min(maxSec, 2 ** (r.failures - after)) * 1000;
|
|
93
|
+
}
|
|
94
|
+
records.delete(key);
|
|
95
|
+
if (records.size >= MAX_RECORDS){
|
|
96
|
+
records.delete(records.keys().next().value); // the longest-idle one
|
|
97
|
+
}
|
|
98
|
+
records.set(key, r);
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
function success(ip, username){
|
|
103
|
+
for (const [, key]of keysOf(ip, username)){
|
|
104
|
+
records.delete(key);
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
return { check, run, failure, success, size: () => records.size };
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
module.exports = { createLoginGuard, POLICIES };
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file packages/coordinator/src/rate-limit.js
|
|
3
|
+
* @description Request rate limiting: token buckets per credential (agent/resource token, dashboard user), per IP without one; a stricter one for sign-in
|
|
4
|
+
*
|
|
5
|
+
* @author Andrian Yablonskyy
|
|
6
|
+
* @copyright Copyright (c) 2026 Andrian Yablonskyy. All rights reserved.
|
|
7
|
+
*
|
|
8
|
+
* This file is part of TestHub and is proprietary and confidential.
|
|
9
|
+
* Unauthorized copying, modification, distribution, or use of this file,
|
|
10
|
+
* via any medium, is strictly prohibited without prior written permission
|
|
11
|
+
* from AdSystem.PRO.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
'use strict';
|
|
15
|
+
|
|
16
|
+
const { hashToken } = require('./services/tokens');
|
|
17
|
+
|
|
18
|
+
// README §12. A token bucket per caller: it holds up to `perMinute` tokens,
|
|
19
|
+
// refills at `perMinute` a minute, and every request takes one — so a short
|
|
20
|
+
// burst (a Client flushing logs, a page loading) goes through while the
|
|
21
|
+
// sustained rate stays capped. Kept in memory: the Coordinator is a single
|
|
22
|
+
// process by design.
|
|
23
|
+
//
|
|
24
|
+
// Who the caller is: the credential when the request carries a valid one
|
|
25
|
+
// (agent token, resource token, dashboard session), else the client IP —
|
|
26
|
+
// unauthenticated calls (sign-in, Client registration), and requests with
|
|
27
|
+
// a token that doesn't exist, so a random token per request can't dodge the
|
|
28
|
+
// limit. Keying by credential means a whole lab behind one NAT address
|
|
29
|
+
// doesn't share a single budget.
|
|
30
|
+
const MAX_BUCKETS = 50_000,
|
|
31
|
+
PRUNE_MS = 60_000;
|
|
32
|
+
|
|
33
|
+
function createBuckets(){
|
|
34
|
+
const buckets = new Map();
|
|
35
|
+
|
|
36
|
+
// Takes one token from `key`'s bucket of `capacity`, refilled at
|
|
37
|
+
// `capacity` per minute. Returns { ok, remaining, retryAfterSec }.
|
|
38
|
+
function take(key, capacity, now = Date.now()){
|
|
39
|
+
const ratePerMs = capacity / 60_000;
|
|
40
|
+
let b = buckets.get(key);
|
|
41
|
+
if (!b){
|
|
42
|
+
if (buckets.size >= MAX_BUCKETS){
|
|
43
|
+
buckets.delete(buckets.keys().next().value); // oldest first
|
|
44
|
+
}
|
|
45
|
+
b = { tokens: capacity, at: now };
|
|
46
|
+
}
|
|
47
|
+
else {
|
|
48
|
+
buckets.delete(key); // re-insert below: Map order = least recently used first
|
|
49
|
+
b.tokens = Math.min(capacity, b.tokens + (now - b.at) * ratePerMs);
|
|
50
|
+
b.at = now;
|
|
51
|
+
}
|
|
52
|
+
buckets.set(key, b);
|
|
53
|
+
if (b.tokens >= 1){
|
|
54
|
+
b.tokens -= 1;
|
|
55
|
+
return { ok: true, remaining: Math.floor(b.tokens), retryAfterSec: 0 };
|
|
56
|
+
}
|
|
57
|
+
return { ok: false, remaining: 0, retryAfterSec: Math.max(1, Math.ceil((1 - b.tokens) / ratePerMs / 1000)) };
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// A bucket that has refilled completely is the same as no bucket.
|
|
61
|
+
function prune(capacityOf, now = Date.now()){
|
|
62
|
+
for (const [key, b]of buckets){
|
|
63
|
+
const capacity = capacityOf(key);
|
|
64
|
+
if (b.tokens + (now - b.at) * (capacity / 60_000) >= capacity){
|
|
65
|
+
buckets.delete(key);
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
return { take, prune, size: () => buckets.size };
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// The credential a request carries, if it's a real one: "agent:<id>",
|
|
74
|
+
// "resource:<id>" or "user:<id>"; otherwise null (counted by IP).
|
|
75
|
+
function credentialOf(req, services){
|
|
76
|
+
const [scheme, token] = (req.get('authorization') || '').split(' ');
|
|
77
|
+
if (scheme === 'Bearer' && token){
|
|
78
|
+
const hash = hashToken(token),
|
|
79
|
+
agent = services.agents.getByTokenHash(hash);
|
|
80
|
+
if (agent){
|
|
81
|
+
return `agent:${agent.id}`;
|
|
82
|
+
}
|
|
83
|
+
const resource = services.registry.getByTokenHash(hash);
|
|
84
|
+
return resource ? `resource:${resource.id}` : null;
|
|
85
|
+
}
|
|
86
|
+
const user = req.session?.user;
|
|
87
|
+
return user ? `user:${user.id}` : null;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function tooMany(req, res, retryAfterSec, what){
|
|
91
|
+
const text = `Too many ${what}: try again in ${retryAfterSec} s.`;
|
|
92
|
+
res.set('Retry-After', String(retryAfterSec));
|
|
93
|
+
if (req.originalUrl.startsWith('/api/') || !req.accepts('html')){
|
|
94
|
+
return res.status(429).json({ error: text, retryAfterSec });
|
|
95
|
+
}
|
|
96
|
+
res.status(429).type('text/plain').send(text);
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
// `config.rateLimit` is read on every request, so a change from the
|
|
100
|
+
// dashboard (Settings, §13.2) applies at once. 0 turns a limit off.
|
|
101
|
+
function createRateLimiter(config, services){
|
|
102
|
+
const buckets = createBuckets(),
|
|
103
|
+
capacityOf = (key) => (key.startsWith('login:') ? config.rateLimit.loginPerMinute : config.rateLimit.requestsPerMinute) || 1,
|
|
104
|
+
pruneTimer = setInterval(() => buckets.prune(capacityOf), PRUNE_MS);
|
|
105
|
+
pruneTimer.unref?.();
|
|
106
|
+
|
|
107
|
+
function limit(req, res, next){
|
|
108
|
+
const perMinute = config.rateLimit.requestsPerMinute,
|
|
109
|
+
loginPerMinute = config.rateLimit.loginPerMinute;
|
|
110
|
+
|
|
111
|
+
// Sign-in attempts: their own, much smaller budget per IP (password
|
|
112
|
+
// guessing), on top of the general one.
|
|
113
|
+
if (loginPerMinute > 0 && req.method === 'POST' && req.path === '/login'){
|
|
114
|
+
const login = buckets.take(`login:${req.ip}`, loginPerMinute);
|
|
115
|
+
if (!login.ok){
|
|
116
|
+
return tooMany(req, res, login.retryAfterSec, 'sign-in attempts');
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
if (!(perMinute > 0)){
|
|
121
|
+
return next();
|
|
122
|
+
}
|
|
123
|
+
const key = credentialOf(req, services) || `ip:${req.ip}`,
|
|
124
|
+
result = buckets.take(key, perMinute);
|
|
125
|
+
res.set({ 'RateLimit-Limit': String(perMinute), 'RateLimit-Remaining': String(result.remaining) });
|
|
126
|
+
if (!result.ok){
|
|
127
|
+
return tooMany(req, res, result.retryAfterSec, 'requests');
|
|
128
|
+
}
|
|
129
|
+
next();
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
limit.buckets = buckets;
|
|
133
|
+
limit.stop = () => clearInterval(pruneTimer);
|
|
134
|
+
return limit;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
module.exports = { createRateLimiter, createBuckets };
|