@wenbin_wb/dsh-bridge 2.10.12 → 2.10.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,11 +1,25 @@
1
- import { spawn, execSync } from 'node:child_process';
2
- import { createWriteStream, createReadStream, existsSync, mkdirSync, readFileSync } from 'node:fs';
1
+ import { spawn, execFileSync } from 'node:child_process';
2
+ import { createWriteStream, createReadStream, existsSync, mkdirSync, readFileSync, statSync, renameSync, writeFileSync } from 'node:fs';
3
3
  import { chmod, stat, unlink, rename } from 'node:fs/promises';
4
4
  import { homedir, platform, arch } from 'node:os';
5
- import { join } from 'node:path';
5
+ import { join, dirname } from 'node:path';
6
6
  import { pipeline } from 'node:stream/promises';
7
7
  import { createHash } from 'node:crypto';
8
8
  import { get as httpsGet } from 'node:https';
9
+ import { get as httpGet } from 'node:http';
10
+
11
+ // 参数化执行外部命令,替代原先的 execSync(`"${path}" --version`) 字符串拼接
12
+ // (CodeQL js/shell-command-constructed-from-input)。execFileSync 不经过 shell,
13
+ // 参数以数组传递,路径里的引号 / $( ) / 反引号不再具备注入语义。
14
+ //
15
+ // Windows 例外:.cmd / .bat 无法被 execFileSync 直接执行(Node 要求经 shell),
16
+ // 官方 cloudflared 在 Windows 是 .exe 走参数化路径,只有测试注入的假二进制是
17
+ // .cmd,故按扩展名决定是否需要 shell。macOS / Linux 恒为 false。
18
+ const IS_WINDOWS = process.platform === 'win32';
19
+ function execBinSync(bin, args, options) {
20
+ const needsShell = IS_WINDOWS && /\.(cmd|bat)$/i.test(bin);
21
+ return execFileSync(bin, args, { ...options, shell: needsShell });
22
+ }
9
23
 
10
24
  const CLOUDFLARED_VERSION = '2024.10.0';
11
25
  const DOWNLOAD_TIMEOUT = 5 * 60 * 1000; // 5 分钟
@@ -15,6 +29,47 @@ const RETRY_BASE_MS = 5 * 1000; // 自愈退避起点 5s
15
29
  const RETRY_MAX_MS = 5 * 60 * 1000; // 自愈退避封顶 5min
16
30
  const DEFAULT_MAX_RETRIES = 12; // 连续失败超过该次数转为 error,不再无限重试
17
31
 
32
+ // ── 运行时健康探针 ────────────────────────────────────────────────────────
33
+ // 事故背景:cloudflared 与 Cloudflare 边缘的连接全部掉光后进程仍存活("半死/假死"),
34
+ // 既不退出也不报错,因此只监听 exit 的旧自愈逻辑永远不会触发,隧道会静默失效
35
+ // (现象:公网 530 + 正文 error code: 1033,持续 2 天无人发现)。
36
+ //
37
+ // 关键约束(一手源码 cloudflared 2024.10.0 `tunnelstate/conntracker.go`):/ready 的
38
+ // readyConnections 取自 CountActiveConns(),而 Disconnected / Reconnecting /
39
+ // RegisteringTunnel / Unregistering 全部把 IsConnected 置 false —— 即
40
+ // **正常重连期间 /ready 同样是 503**。因此单次 503 无法区分"连接器假死"与
41
+ // "网络暂时全断、cloudflared 正按自身退避重连"。唯一能区分的是"持续多久",
42
+ // 故采用两级升级,且强制重建永不进入终态(见 _scheduleRestart 的 terminal 选项)。
43
+ const HEALTH_PROBE_INTERVAL_MS = 30 * 1000; // 探活间隔
44
+ const HEALTH_PROBE_TIMEOUT_MS = 5 * 1000; // 单次探活总时限(远程调用防失控)
45
+ const HEALTH_DEGRADED_THRESHOLD = 3; // 连续失败达此数 → 面板可见降级(约 90s 发现)
46
+ const HEALTH_RESTART_THRESHOLD = 10; // 连续失败达此数 → 终止进程强制重建(约 5min)
47
+
48
+ // ── cloudflared 输出落盘 ──────────────────────────────────────────────────
49
+ // 事故时 journal 里查不到任何 cloudflared 日志,导致事后无法定位断连时间点与原因。
50
+ export const CLOUDFLARED_LOG_NAME = 'cloudflared.log';
51
+ const CLOUDFLARED_LOG_MAX_BYTES = 5 * 1024 * 1024; // 超限轮转为 .1(覆盖旧的)
52
+
53
+ // 从 cloudflared 自报日志中解析 metrics 服务地址(运行时健康探针的唯一入口)。
54
+ // 真实输出形如:INF Starting metrics server on 127.0.0.1:41143/metrics
55
+ // 解析不到 → 显式告警并跳过探活(绝不静默降级,也绝不因探测不到而误杀进程)。
56
+ //
57
+ // 必须在这里就把非法地址拦掉:node:http 的 get() 对畸形 URL 会**同步抛错**
58
+ // (ERR_INVALID_URL),异常会穿透探针 Promise 造成未处理 rejection ——
59
+ // Node 默认 unhandled-rejections=throw 会直接结束整个 DSH 主进程,
60
+ // 也就是说"本该保护服务的探针"反而会把服务干掉。端口必须落在 1..65535。
61
+ export function parseMetricsAddress(text) {
62
+ if (!text) return null;
63
+ const m = /Starting metrics server on (\S+?)\/metrics/.exec(text);
64
+ if (!m) return null;
65
+ const addr = m[1];
66
+ const parts = /^([A-Za-z0-9._-]+|\[[0-9A-Fa-f:]+\]):(\d{1,5})$/.exec(addr);
67
+ if (!parts) return null;
68
+ const port = Number(parts[2]);
69
+ if (!Number.isInteger(port) || port < 1 || port > 65535) return null;
70
+ return addr;
71
+ }
72
+
18
73
  // ── 确定性失败特征:cloudflared 因配置/用法错误退出(非网络瞬态)─────────
19
74
  // 这类失败重试无意义,识别后直接置 error 让用户看到明确原因,而不是
20
75
  // 误判成"意外退出"退避重连 N 次(issue #35 作者建议)。
@@ -116,7 +171,7 @@ function findSystemCloudflared() {
116
171
  if (bin.includes('/') || bin.includes('\\')) {
117
172
  if (!existsSync(bin)) continue;
118
173
  }
119
- execSync(`"${bin}" --version`, { stdio: 'ignore', timeout: 3000 });
174
+ execBinSync(bin, ['--version'], { stdio: 'ignore', timeout: 3000 });
120
175
  return bin;
121
176
  } catch {}
122
177
  }
@@ -138,10 +193,26 @@ export class CloudflaredManager {
138
193
  * @param {number} [opts.handshakeTimeoutMs=90000] 等待隧道就绪的握手超时(测试可注入小值)。
139
194
  * @param {object} [opts.spawnOptions] 透传给 child_process.spawn 的额外选项(测试注入用,
140
195
  * 如 Windows 下需 shell:true 才能运行 .cmd mock;生产不传)。
196
+ * @param {number} [opts.healthProbeIntervalMs=30000] 就绪后运行时探活间隔。
197
+ * @param {number} [opts.healthProbeTimeoutMs=5000] 单次探活总时限。
198
+ * @param {number} [opts.healthDegradedThreshold=3] 连续失败达此数即在面板上显示降级
199
+ * (默认约 90s——这是"发现故障"的时点)。
200
+ * @param {number} [opts.healthRestartThreshold=10] 连续失败达此数才终止进程强制重建
201
+ * (默认约 5min)。必须显著大于 degraded 阈值:/ready 无法区分"假死"与
202
+ * "正在自行重连",只有持续时长能区分,过早重建会误杀本可自愈的连接器。
203
+ * @param {string|null} [opts.logFilePath=null] cloudflared stdout/stderr 落盘路径;
204
+ * null 表示不落盘(默认)。生产由调用方显式传入(见 lib/index.js),
205
+ * 避免库的默认行为在单测里往真实运维日志目录写测试内容。
206
+ * @param {number} [opts.logMaxBytes=5242880] 单文件超限后轮转为 `<logFilePath>.1`。
141
207
  */
142
208
  constructor({ port, home, token, hostname, onStateChange, logger,
143
209
  binaryPath, retryPolicy, noAutoupdate = true, binaryVersion = CLOUDFLARED_VERSION,
144
- handshakeTimeoutMs = HANDSHAKE_TIMEOUT_MS, spawnOptions = null }) {
210
+ handshakeTimeoutMs = HANDSHAKE_TIMEOUT_MS, spawnOptions = null,
211
+ healthProbeIntervalMs = HEALTH_PROBE_INTERVAL_MS,
212
+ healthProbeTimeoutMs = HEALTH_PROBE_TIMEOUT_MS,
213
+ healthDegradedThreshold = HEALTH_DEGRADED_THRESHOLD,
214
+ healthRestartThreshold = HEALTH_RESTART_THRESHOLD,
215
+ logFilePath = null, logMaxBytes = CLOUDFLARED_LOG_MAX_BYTES }) {
145
216
  this.port = port;
146
217
  this.home = home || join(homedir(), '.dsh-bridge');
147
218
  this.token = token ? String(token).trim() : null;
@@ -170,18 +241,42 @@ export class CloudflaredManager {
170
241
  // 测试注入的 spawn 选项(如 Windows 的 shell);生产为 null 不影响默认行为
171
242
  this._spawnOptions = spawnOptions || null;
172
243
 
244
+ // 运行时健康探针参数(测试可注入小值)
245
+ this.healthProbeIntervalMs = healthProbeIntervalMs;
246
+ this.healthProbeTimeoutMs = healthProbeTimeoutMs;
247
+ this.healthDegradedThreshold = healthDegradedThreshold;
248
+ this.healthRestartThreshold = healthRestartThreshold;
249
+ this.logFilePath = logFilePath || null;
250
+ this.logMaxBytes = logMaxBytes;
251
+
173
252
  this.process = null;
174
253
  this.url = null;
175
254
  this.binaryPath = null;
176
255
  this._stopped = false;
177
256
  this._retryTimer = null;
178
257
  this._restartCount = 0; // 连续启动失败/意外退出次数,就绪后清零
258
+
259
+ this._metricsAddress = null; // 本次 spawn 的 cloudflared metrics 地址(探活入口)
260
+ this._metricsParseTail = ''; // metrics 日志行的跨 chunk 拼接缓冲
261
+ this._probeTimer = null;
262
+ this._probeFailures = 0;
263
+ this._probeInFlight = false; // 防重叠探活堆积
264
+ this._probeCount = 0; // 已完成的探活次数(供测试断言探针确实在跑)
265
+ this._probeLogFailures = 0; // 探针日志通道自身抛错的次数(供自查是否丢过日志)
266
+ this._spawnSeq = 0; // spawn 代数:探活结果据此作废,避免算到新进程头上
267
+ this._healthRecovering = false; // 健康探针触发的重建链路中(此期间不允许终态)
268
+ this._readyState = null; // 最近一次 ready 状态,供探活恢复后回写
269
+ this._logStream = null;
270
+ this._logBytesWritten = 0;
271
+ this._redactCarry = ''; // 跨 chunk 的脱敏残留(防 Token 被切分而漏网)
179
272
  }
180
273
 
181
274
  // 异步启动,立即返回——调用方不需要 await
182
275
  start() {
183
276
  this._stopped = false;
184
277
  this._restartCount = 0;
278
+ this._healthRecovering = false;
279
+ this._probeInFlight = false;
185
280
  if (this._retryTimer) {
186
281
  clearTimeout(this._retryTimer);
187
282
  this._retryTimer = null;
@@ -206,6 +301,9 @@ export class CloudflaredManager {
206
301
  }
207
302
 
208
303
  // 退避自愈调度:唯一入口在"启动失败"与"就绪后意外退出"。stop()/超限 终止。
304
+ // _healthRecovering 期间(健康探针判定假死后触发的整条重建链路,含重启后握手失败)
305
+ // 永不进入终态——必须覆盖整条链路而不只是那次 kill:重启时网络往往仍不通,
306
+ // 新进程握手失败仍会走这里,若在此处转 error 就把"可自愈的断网"变成人工故障。
209
307
  _scheduleRestart(reason) {
210
308
  if (this._stopped) return; // 用户已停止,绝不复活
211
309
  if (this._retryTimer) return; // 已在倒计时中,避免叠加调度
@@ -215,15 +313,25 @@ export class CloudflaredManager {
215
313
  }
216
314
  this._restartCount++;
217
315
  if (this._restartCount > this.retry.maxRetries) {
218
- this._setState('error', `${reason}(已自动重试 ${this.retry.maxRetries} 次仍失败,请检查网络/Token,或点击「关闭」停止)`);
219
- return;
316
+ if (!this._healthRecovering) {
317
+ this._setState('error', `${reason}(已自动重试 ${this.retry.maxRetries} 次仍失败,请检查网络/Token,或点击「关闭」停止)`);
318
+ return;
319
+ }
320
+ // 健康探针触发的重建链路永不进入终态:长时断网时 cloudflared 自身也是无限退避
321
+ // 重连,这里保持同样语义。退避仍随 _restartCount 增长并封顶 maxDelayMs,
322
+ // 不会退化成高频重启风暴;重新就绪(tryResolve)后该标记自动清除。
220
323
  }
221
324
  const delay = Math.min(
222
325
  this.retry.baseDelayMs * 2 ** (this._restartCount - 1),
223
326
  this.retry.maxDelayMs
224
327
  );
225
- this._setState('reconnecting',
226
- `${reason},${Math.max(1, Math.round(delay / 1000))}s 后自动重连(第 ${this._restartCount}/${this.retry.maxRetries} 次)`);
328
+ // 文案必须如实:健康重建链路会故意突破 maxRetries 持续重试,
329
+ // 若沿用"第 N/M 次"会出现"第 9/2 次"这种自相矛盾的提示,反而误导排障。
330
+ const delayText = delay >= 1000 ? `${Math.round(delay / 1000)}s` : `${delay}ms`;
331
+ const beyondCap = this._restartCount > this.retry.maxRetries;
332
+ this._setState('reconnecting', beyondCap
333
+ ? `${reason},${delayText} 后继续自动重连(已重试 ${this._restartCount} 次;隧道长时间无法建立,请检查网络与 Token)`
334
+ : `${reason},${delayText} 后自动重连(第 ${this._restartCount}/${this.retry.maxRetries} 次)`);
227
335
  this._retryTimer = setTimeout(() => {
228
336
  this._retryTimer = null;
229
337
  this._restartAttempt();
@@ -261,7 +369,7 @@ export class CloudflaredManager {
261
369
  // 校验自管理二进制是否匹配期望版本;系统级二进制不校验(尊重用户安装)
262
370
  _checkManagedBinaryVersion(binPath) {
263
371
  try {
264
- const out = execSync(`"${binPath}" --version`, { encoding: 'utf8', timeout: 3000 });
372
+ const out = execBinSync(binPath, ['--version'], { encoding: 'utf8', timeout: 3000 });
265
373
  const ver = parseCloudflaredVersion(out);
266
374
  if (!ver) return { ok: false, reason: `无法解析版本输出: ${(out || '').trim().slice(0, 80)}` };
267
375
  if (ver !== this.binaryVersion) {
@@ -309,7 +417,7 @@ export class CloudflaredManager {
309
417
  if (platform() !== 'win32') {
310
418
  await chmod(binPath, 0o755).catch(() => {});
311
419
  if (platform() === 'darwin') {
312
- try { execSync(`xattr -d com.apple.quarantine "${binPath}"`, { stdio: 'ignore' }); } catch {}
420
+ try { execBinSync('xattr', ['-d', 'com.apple.quarantine', binPath], { stdio: 'ignore' }); } catch {}
313
421
  }
314
422
  }
315
423
  // 版本钉死校验:cloudflared autoupdate 可能已把钉死的版本自替换成新版,
@@ -345,7 +453,7 @@ export class CloudflaredManager {
345
453
 
346
454
  if (url.endsWith('.tgz') || url.endsWith('.tar.gz')) {
347
455
  try {
348
- execSync(`tar -xzf "${tempPath}" -C "${binDir}"`);
456
+ execBinSync('tar', ['-xzf', tempPath, '-C', binDir]);
349
457
  await unlink(tempPath).catch(() => {});
350
458
  } catch (tarErr) {
351
459
  this.logger?.error('解压 cloudflared 压缩包失败: %s', tarErr.message);
@@ -359,7 +467,7 @@ export class CloudflaredManager {
359
467
  if (platform() !== 'win32') {
360
468
  await chmod(binPath, 0o755).catch(() => {});
361
469
  if (platform() === 'darwin') {
362
- try { execSync(`xattr -d com.apple.quarantine "${binPath}"`, { stdio: 'ignore' }); } catch {}
470
+ try { execBinSync('xattr', ['-d', 'com.apple.quarantine', binPath], { stdio: 'ignore' }); } catch {}
363
471
  }
364
472
  }
365
473
 
@@ -402,6 +510,14 @@ export class CloudflaredManager {
402
510
  });
403
511
  this.process = proc;
404
512
 
513
+ // 本次 spawn 独立状态:metrics 地址与日志流都随进程生命周期重建;
514
+ // 代数(_spawnSeq)用于让飞行中的旧探活结果作废——重启后 cloudflared 很
515
+ // 可能复用同一个 metrics 端口,仅比对地址字符串不足以识别"这是新进程"。
516
+ this._metricsAddress = null;
517
+ this._metricsParseTail = '';
518
+ this._spawnSeq++;
519
+ this._openLogStream();
520
+
405
521
  let resolved = false;
406
522
  let timeoutTimer = null;
407
523
  // 累积 stderr 尾部(供 exit 时判断确定性失败:CLI 用法错误 / 配置错误)
@@ -418,8 +534,10 @@ export class CloudflaredManager {
418
534
  clearTimeout(timeoutTimer);
419
535
  timeoutTimer = null;
420
536
  }
421
- // 就绪即证明链路可用,连续失败计数清零
537
+ // 就绪即证明链路可用:连续失败计数清零,并退出"健康重建"模式
422
538
  this._restartCount = 0;
539
+ this._healthRecovering = false;
540
+ this._startHealthProbe(); // 就绪后才探活:启动阶段的问题由握手超时/exit 负责
423
541
  resolve();
424
542
  }
425
543
  };
@@ -460,10 +578,34 @@ export class CloudflaredManager {
460
578
  }
461
579
  };
462
580
 
463
- proc.stdout.on('data', (d) => parseUrl(d.toString()));
581
+ // metrics 地址必须先于就绪文本被记录:cloudflared 是先起 metrics 服务再注册连接。
582
+ // 两处稳健性处理:
583
+ // 1) 该日志行本身也可能被 chunk 切断 → 用滚动缓冲拼接后再匹配;
584
+ // 2) 地址可能"迟到"(就绪时还没解析到)→ 一旦补上就补启探活,否则安全网永久不启用。
585
+ const captureMetricsAddress = (text) => {
586
+ if (this._metricsAddress) return;
587
+ const buffered = (this._metricsParseTail + text).slice(-512);
588
+ this._metricsParseTail = buffered;
589
+ const addr = parseMetricsAddress(buffered);
590
+ if (!addr) return;
591
+ this._metricsAddress = addr;
592
+ if (resolved && !this._stopped && !this._probeTimer) {
593
+ this._probeLog('info', 'metrics 地址在就绪后才解析到,补启运行时健康探活: %s', addr);
594
+ this._startHealthProbe();
595
+ }
596
+ };
597
+
598
+ proc.stdout.on('data', (d) => {
599
+ const text = d.toString();
600
+ this._appendCloudflaredLog(text);
601
+ captureMetricsAddress(text);
602
+ parseUrl(text);
603
+ });
464
604
  proc.stderr.on('data', (d) => {
465
605
  const text = d.toString();
466
606
  this.logger?.debug('cloudflared: %s', text.trim());
607
+ this._appendCloudflaredLog(text);
608
+ captureMetricsAddress(text);
467
609
  stderrTail = (stderrTail + text).slice(-2000); // 只保留尾部 2KB
468
610
  parseUrl(text);
469
611
  if (text.includes('Registered tunnel') && !resolved) {
@@ -476,6 +618,9 @@ export class CloudflaredManager {
476
618
  clearTimeout(timeoutTimer);
477
619
  timeoutTimer = null;
478
620
  }
621
+ // 探活定时器与日志流都绑定本次 spawn,退出即回收(自愈会重新建立)
622
+ this._stopHealthProbe();
623
+ this._closeLogStream();
479
624
  // exit 时"仍是当前进程"才允许清理引用 + 调度自愈。
480
625
  // 自愈已 spawn 新进程后,旧进程迟到的 exit 不满足 stillCurrent → 静默,
481
626
  // 避免"新进程连接中、旧 exit 又触发一次重启"的竞态(否则会叠出第三进程)。
@@ -491,7 +636,8 @@ export class CloudflaredManager {
491
636
  if (isFatalCloudflaredError(stderrTail)) err.fatal = true;
492
637
  reject(err);
493
638
  } else if (!this._stopped && stillCurrent) {
494
- // 就绪后的意外退出(崩溃 / OOM / 误杀 / autoupdate 残留自替换)→ 退避自愈
639
+ // 就绪后的意外退出(崩溃 / OOM / 误杀 / autoupdate 残留自替换 / 健康探针判定假死)
640
+ // → 退避自愈。是否允许终态由 _scheduleRestart 依 _healthRecovering 判定。
495
641
  this._scheduleRestart(`cloudflared 进程意外退出 (code=${code ?? ''}${signal ? `, ${signal}` : ''})`);
496
642
  } else {
497
643
  this._setState('idle', '');
@@ -503,6 +649,8 @@ export class CloudflaredManager {
503
649
  clearTimeout(timeoutTimer);
504
650
  timeoutTimer = null;
505
651
  }
652
+ this._stopHealthProbe();
653
+ this._closeLogStream();
506
654
  if (!resolved) reject(err);
507
655
  });
508
656
 
@@ -518,7 +666,298 @@ export class CloudflaredManager {
518
666
  });
519
667
  }
520
668
 
521
- _setState(phase, detail) {
669
+ // ── cloudflared 输出落盘(含 Token 脱敏与大小轮转)──────────────────────
670
+
671
+ // Token 属敏感凭据,落盘前一律替换;不依赖上游是否会自行脱敏。
672
+ //
673
+ // 关键:stderr 的 'data' 事件不保证按行对齐,Token 可能被切分在两个块之间。
674
+ // 这里用**逐字符扫描**而不是"整块 split 替换":
675
+ // - 任意位置命中完整 Token → 替换为 ***(完整 Token 绝不会明文落盘);
676
+ // - 只在"剩余不足一个 Token 长度、且恰好是 Token 前缀"时才扣住,等下一块补齐。
677
+ // 不能退化成"扣掉最长的 Token 前缀后缀"这种简化写法:若 Token 自身存在 border
678
+ // (例如首尾字符相同),完整 Token 结尾处也会被误判成前缀而扣下一截,结果前半段
679
+ // 与扣下的那截先后写出、在文件里重新拼成完整 Token(已实测的泄漏路径)。
680
+ _redactToken(text) {
681
+ if (!this.token) return text;
682
+ const token = this.token;
683
+ const L = token.length;
684
+ const combined = this._redactCarry + text;
685
+ this._redactCarry = '';
686
+ let out = '';
687
+ let i = 0;
688
+ while (i < combined.length) {
689
+ if (i + L <= combined.length && combined.startsWith(token, i)) {
690
+ out += '***';
691
+ i += L;
692
+ continue;
693
+ }
694
+ if (i + L > combined.length) {
695
+ const rest = combined.slice(i);
696
+ if (token.startsWith(rest)) {
697
+ this._redactCarry = rest; // 可能是被切开的 Token 前半段
698
+ break;
699
+ }
700
+ }
701
+ out += combined[i];
702
+ i++;
703
+ }
704
+ return out;
705
+ }
706
+
707
+ _openLogStream() {
708
+ this._closeLogStream(); // 幂等:自愈重启时先关掉上一个进程的流,避免句柄堆积
709
+ if (!this.logFilePath) return;
710
+ try {
711
+ mkdirSync(dirname(this.logFilePath), { recursive: true });
712
+ // 同步把文件建出来:createWriteStream 是异步 open,若不在此时落地,
713
+ // 紧随其后的轮转判定会遇到"文件还没存在"而 rename 落空(运行中轮转失效)。
714
+ writeFileSync(this.logFilePath, '', { flag: 'a' });
715
+ this._rotateIfOversize();
716
+ this._logBytesWritten = existsSync(this.logFilePath) ? statSync(this.logFilePath).size : 0;
717
+ } catch (err) {
718
+ // 目录创建/建文件/取大小失败:不阻断日志写入,但必须让问题可见
719
+ this.logger?.warn('cloudflared 日志目录或文件准备失败(本次可能不落盘): %s', err.message);
720
+ }
721
+ try {
722
+ const stream = createWriteStream(this.logFilePath, { flags: 'a' });
723
+ stream.on('error', (err) => {
724
+ // 流已不可写:置空引用,避免后续继续往死流写并重复刷同一条告警
725
+ this._logStream = null;
726
+ this.logger?.warn('cloudflared 日志写入失败: %s', err.message);
727
+ });
728
+ this._logStream = stream;
729
+ } catch (err) {
730
+ this._logStream = null;
731
+ this.logger?.warn('无法打开 cloudflared 日志文件 %s: %s', this.logFilePath, err.message);
732
+ }
733
+ }
734
+
735
+ _rotateIfOversize() {
736
+ if (!this.logFilePath) return;
737
+ try {
738
+ if (existsSync(this.logFilePath) && statSync(this.logFilePath).size >= this.logMaxBytes) {
739
+ renameSync(this.logFilePath, `${this.logFilePath}.1`);
740
+ this._logBytesWritten = 0;
741
+ }
742
+ } catch (err) {
743
+ this.logger?.warn('cloudflared 日志轮转失败(继续追加写): %s', err.message);
744
+ }
745
+ }
746
+
747
+ _closeLogStream() {
748
+ // 无论流是否可用,carry 都必须清空——否则残留的 Token 片段会污染下一个
749
+ // 进程的第一行日志(流先报 error 置空、随后 close 早退的场景实测可复现)。
750
+ const carry = this._redactCarry;
751
+ this._redactCarry = '';
752
+ if (!this._logStream) return;
753
+ // 流关闭时仍扣着的必然是 Token 的一个前缀(见 _redactToken):以占位符收尾。
754
+ // 取舍说明:这会牺牲最多 token.length-1 个正常字符的日志保真度,
755
+ // 但换来"任何 Token 片段都不落盘"。极端情况下宁可少一行尾巴,不可留密钥片段。
756
+ if (carry) this._logStream.write('***');
757
+ this._logStream.end();
758
+ this._logStream = null;
759
+ }
760
+
761
+ _appendCloudflaredLog(text) {
762
+ if (!this._logStream || !text) return;
763
+ const out = this._redactToken(text);
764
+ if (!out) return;
765
+ const bytes = Buffer.byteLength(out, 'utf8');
766
+ // 运行中也检查轮转:只在 spawn 时检查会让长寿命进程把日志写爆。
767
+ // 这里按"本次运行已写入字节数"判定,不依赖 statSync——写流是异步的,
768
+ // 刚写完就 stat 可能仍是旧大小,会造成轮转判定抖动。
769
+ if (this._logBytesWritten + bytes >= this.logMaxBytes) {
770
+ this._closeLogStream();
771
+ this._renameToRotated();
772
+ this._openLogStream();
773
+ if (!this._logStream) return;
774
+ }
775
+ this._logStream.write(out);
776
+ this._logBytesWritten += bytes;
777
+ }
778
+
779
+ _renameToRotated() {
780
+ if (!this.logFilePath) return;
781
+ try {
782
+ if (existsSync(this.logFilePath)) renameSync(this.logFilePath, `${this.logFilePath}.1`);
783
+ } catch (err) {
784
+ this.logger?.warn('cloudflared 日志轮转失败(继续追加写): %s', err.message);
785
+ }
786
+ this._logBytesWritten = 0;
787
+ }
788
+
789
+ // ── 运行时健康探针 ──────────────────────────────────────────────────────
790
+
791
+ // 就绪后启动探活。metrics 地址解析失败时显式告警并跳过——绝不静默降级,
792
+ // 也绝不在没有可信探活通道的情况下误杀进程;同时把降级写进面板可见的状态详情,
793
+ // 否则"整个安全网失效"只会是一行没人看的 journal warning(本次事故的教训)。
794
+ // 探针专用安全日志:探针跑在定时器与子进程数据回调里,异常没有任何上层接住,
795
+ // 一旦 logger 自身抛错就会变成未处理 rejection —— Node 默认 unhandled-rejections=throw
796
+ // 会直接结束宿主进程,即"本该保护服务的探针把服务干掉"。故此处绝不允许抛出。
797
+ _probeLog(level, msg, ...args) {
798
+ try {
799
+ this.logger?.[level]?.(msg, ...args);
800
+ } catch (logErr) {
801
+ // 日志通道本身故障:此处再抛就会拖垮宿主进程,故刻意吞掉。
802
+ // 这不是"静默降级"——要输出的降级信息本身就在这条已损坏的通道上,无处可写;
803
+ // 且关键降级仍经 _setState → 面板独立可见,不依赖 logger。
804
+ // 注意:该计数目前只在单测中被读取,生产尚无出口(后续可接入面板/统计),
805
+ // 因此不要把它当作"生产可自查"的手段。
806
+ this._probeLogFailures++;
807
+ }
808
+ }
809
+
810
+ _startHealthProbe() {
811
+ this._stopHealthProbe();
812
+ if (!this._metricsAddress) {
813
+ const detail = `${this._readyState?.detail || '隧道已建立'};⚠️ 运行时健康探活未启用`
814
+ + '(未识别到 cloudflared metrics 地址),连接器若静默假死将无法自动发现';
815
+ this._probeLog('warn', '未从 cloudflared 输出识别到 metrics 地址,本次运行跳过健康探活'
816
+ + '(连接器若静默假死将无法自动发现,请检查 cloudflared 版本是否变更了 metrics 日志文案)');
817
+ // record:false —— 这是降级提示,不能污染"纯净的 ready 详情",
818
+ // 否则地址迟到后补启探活时将无法把面板文案恢复成正常状态。
819
+ this._setState('ready', detail, { record: false });
820
+ return;
821
+ }
822
+ // 地址迟到后补启探活:把面板文案恢复为正常的 ready 详情
823
+ if (this._readyState) this._setState(this._readyState.phase, this._readyState.detail);
824
+ this._probeTimer = setInterval(() => {
825
+ // 必须显式 catch:探针里任何逃逸的异常(如畸形 metrics 地址让 http.get 同步抛错)
826
+ // 都会变成未处理 rejection,而 Node 默认 unhandled-rejections=throw 会直接
827
+ // 结束整个 DSH 主进程 —— 那正是探针本该防止的"服务整体消失"。
828
+ // 用 _probeLog 而非 logger 直调:catch 处理器自身也不允许再抛。
829
+ this._probeOnce().catch((err) => {
830
+ this._probeLog('warn', '健康探活出现未预期异常(已忽略,不影响连接器运行): %s', err?.message ?? err);
831
+ });
832
+ }, this.healthProbeIntervalMs);
833
+ this._probeTimer.unref?.(); // 不因探活定时器而阻止进程退出
834
+ }
835
+
836
+ _stopHealthProbe() {
837
+ if (this._probeTimer) {
838
+ clearInterval(this._probeTimer);
839
+ this._probeTimer = null;
840
+ }
841
+ this._probeFailures = 0;
842
+ }
843
+
844
+ async _probeOnce() {
845
+ // 防重叠:单次探活最长 healthProbeTimeoutMs,若上一轮还没结论就不再叠加请求
846
+ if (this._probeInFlight) return;
847
+ const addr = this._metricsAddress;
848
+ if (!addr) return;
849
+ const seq = this._spawnSeq; // spawn 代数:据此识别"结果属于哪个进程"
850
+
851
+ this._probeInFlight = true;
852
+ let result;
853
+ try {
854
+ result = await this._checkTunnelReady(addr);
855
+ } catch (err) {
856
+ // 兜底:探针内部任何异常都不得向外逃逸(见 setInterval 处的说明)
857
+ result = { healthy: false, reason: `探针内部异常: ${err?.message ?? err}` };
858
+ } finally {
859
+ this._probeInFlight = false;
860
+ }
861
+ this._probeCount++;
862
+
863
+ // 探活期间可能已被 stop()、或已自愈重启到新进程 → 结果作废。
864
+ // 必须比对代数而非地址:重启后 cloudflared 很可能复用同一个 metrics 端口。
865
+ if (this._stopped || this._spawnSeq !== seq) return;
866
+
867
+ if (result.healthy) {
868
+ if (this._probeFailures >= this.healthDegradedThreshold) {
869
+ this._probeLog('info', '隧道健康探活已自行恢复(此前连续失败 %d 次)', this._probeFailures);
870
+ if (this._readyState) this._setState(this._readyState.phase, this._readyState.detail);
871
+ }
872
+ this._probeFailures = 0;
873
+ return;
874
+ }
875
+
876
+ this._probeFailures++;
877
+ const fails = this._probeFailures;
878
+ const why = result.reason || '未知原因';
879
+
880
+ if (fails === this.healthDegradedThreshold) {
881
+ // 第一级:可见降级。此时不杀进程——/ready 为 503 也可能只是 cloudflared
882
+ // 正在按自身退避重连,过早介入反而会把本可自愈的故障变成人工故障。
883
+ this._probeLog('warn', '隧道健康探活连续 %d 次失败(%s):连接器可能已无健康边缘连接,'
884
+ + '转入降级观察,达 %d 次将强制重建', fails, why, this.healthRestartThreshold);
885
+ this._setState('reconnecting', `健康探活连续 ${fails} 次失败:连接器已无健康边缘连接,`
886
+ + `正在观察能否自行恢复(${fails}/${this.healthRestartThreshold})`);
887
+ } else if (fails < this.healthDegradedThreshold) {
888
+ this._probeLog('warn', '隧道健康探活失败(第 %d/%d 次): %s;原因: %s',
889
+ fails, this.healthRestartThreshold, addr, why);
890
+ } else {
891
+ this._probeLog('warn', '隧道健康探活仍失败(第 %d/%d 次): %s', fails, this.healthRestartThreshold, why);
892
+ }
893
+
894
+ if (fails < this.healthRestartThreshold) return;
895
+
896
+ // 第二级:持续不恢复 → 判定假死,强制重建。
897
+ // 置 _healthRecovering:整条重建链路(含重启后握手失败)都不允许进入终态。
898
+ this._probeLog('error', '隧道连接器连续 %d 次探活失败(约 %d 分钟):判定为假死,'
899
+ + '终止进程以触发重建;该重建不设终态上限,网络恢复后仍会自愈',
900
+ fails, Math.round(fails * this.healthProbeIntervalMs / 60000));
901
+ this._stopHealthProbe();
902
+ this._healthRecovering = true;
903
+ this._terminateProcess(); // 不自行重启:交给 exit → _scheduleRestart 既有链路
904
+ }
905
+
906
+ // 探活判据:HTTP 200 且 readyConnections > 0(与 cloudflared 官方 /ready 语义一致)。
907
+ // 返回 { healthy, reason }:reason 会进入 warn 级日志,生产关闭 debug 时也能看到原因。
908
+ // 注意:http 的 timeout 选项只是 **socket 不活动** 超时,不是总时限——对"持续滴流
909
+ // 字节"的响应永不触发,因此这里额外加一个绝对截止时间兜底。
910
+ _checkTunnelReady(addr) {
911
+ return new Promise((resolve) => {
912
+ let settled = false;
913
+ let deadline = null;
914
+ let req = null;
915
+ const done = (healthy, reason) => {
916
+ if (settled) return;
917
+ settled = true;
918
+ if (deadline) clearTimeout(deadline);
919
+ try { req?.destroy(); } catch { /* 已结束的请求 destroy 无害 */ }
920
+ resolve({ healthy, reason });
921
+ };
922
+
923
+ deadline = setTimeout(() => {
924
+ done(false, `探活超过总时限 ${this.healthProbeTimeoutMs}ms`);
925
+ }, this.healthProbeTimeoutMs);
926
+
927
+ // httpGet 对畸形 URL 会同步抛错;虽然 parseMetricsAddress 已在入口校验,
928
+ // 这里再兜一层,确保该函数永不向外抛(否则会变成未处理 rejection 拖垮主进程)。
929
+ try {
930
+ req = httpGet(`http://${addr}/ready`, { timeout: this.healthProbeTimeoutMs }, (res) => {
931
+ let body = '';
932
+ res.setEncoding('utf8');
933
+ res.on('data', (chunk) => { body += chunk; });
934
+ res.on('end', () => {
935
+ if (res.statusCode !== 200) {
936
+ done(false, `HTTP ${res.statusCode}`);
937
+ return;
938
+ }
939
+ try {
940
+ const parsed = JSON.parse(body);
941
+ const conns = Number(parsed?.readyConnections);
942
+ done(conns > 0, conns > 0 ? '' : `HTTP 200 但 readyConnections=${parsed?.readyConnections}`);
943
+ } catch (err) {
944
+ done(false, `响应无法解析为 JSON: ${err.message}`);
945
+ }
946
+ });
947
+ });
948
+ } catch (err) {
949
+ done(false, `无法发起探活请求(地址可能非法): ${err.message}`);
950
+ return;
951
+ }
952
+ req.on('timeout', () => { done(false, `socket 不活动超过 ${this.healthProbeTimeoutMs}ms`); });
953
+ req.on('error', (err) => { done(false, `请求失败: ${err.message}`); });
954
+ });
955
+ }
956
+
957
+ _setState(phase, detail, { record = true } = {}) {
958
+ // 记住最近一次"纯净"的 ready 状态:探活由降级恢复、或地址迟到后补启探活时
959
+ // 要回写它,避免面板长期停留在降级文案上(record:false 用于降级提示本身)。
960
+ if (phase === 'ready' && record) this._readyState = { phase, detail };
522
961
  this.onStateChange?.({ phase, detail });
523
962
  }
524
963
 
@@ -529,6 +968,8 @@ export class CloudflaredManager {
529
968
  this._retryTimer = null;
530
969
  }
531
970
  this._restartCount = 0;
971
+ this._stopHealthProbe();
972
+ this._closeLogStream();
532
973
  if (this.process) {
533
974
  this.logger?.info('停止 cloudflared...');
534
975
  this._terminateProcess();