omp-conductor 0.15.9 → 0.15.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/README.md +273 -2543
  2. package/REFERENCE.md +2638 -0
  3. package/package.json +3 -2
  4. package/schema/config.schema.json +8 -23
  5. package/src/arm-challenge.ts +112 -0
  6. package/src/ask.ts +434 -0
  7. package/src/board.ts +81 -15
  8. package/src/brief-upgrade.ts +114 -8
  9. package/src/briefs/orchestrator.md +55 -29
  10. package/src/briefs/policy.md +14 -5
  11. package/src/briefs/worker.md +7 -1
  12. package/src/chain-check.ts +1 -1
  13. package/src/check-trailing-newlines.ts +82 -0
  14. package/src/cli.ts +190 -1391
  15. package/src/commands/arm.ts +21 -0
  16. package/src/commands/board.ts +23 -0
  17. package/src/commands/brief-upgrade.ts +186 -0
  18. package/src/commands/context.ts +49 -0
  19. package/src/commands/daemon.ts +71 -0
  20. package/src/commands/dashboard.ts +74 -0
  21. package/src/commands/decision.ts +103 -0
  22. package/src/commands/disarm.ts +21 -0
  23. package/src/commands/doctor.ts +98 -0
  24. package/src/commands/event.ts +62 -0
  25. package/src/commands/extend.ts +64 -0
  26. package/src/commands/friction.ts +56 -0
  27. package/src/commands/help.ts +9 -0
  28. package/src/commands/hold.ts +26 -0
  29. package/src/commands/intake.ts +134 -0
  30. package/src/commands/ledger.ts +69 -0
  31. package/src/commands/message.ts +48 -0
  32. package/src/commands/report.ts +170 -0
  33. package/src/commands/restart.ts +76 -0
  34. package/src/commands/resume.ts +58 -0
  35. package/src/commands/setup.ts +93 -0
  36. package/src/commands/start.ts +23 -0
  37. package/src/commands/stats.ts +131 -0
  38. package/src/commands/status.ts +48 -0
  39. package/src/commands/stop.ts +51 -0
  40. package/src/commands/tail.ts +109 -0
  41. package/src/commands/unblock.ts +39 -0
  42. package/src/commands/upgrade-install.ts +31 -0
  43. package/src/commands/upgrade-rollback.ts +23 -0
  44. package/src/commands/upgrade.ts +25 -0
  45. package/src/commands/verb.ts +83 -0
  46. package/src/commands/version.ts +30 -0
  47. package/src/commands/worker.ts +100 -0
  48. package/src/config-schema.ts +38 -1
  49. package/src/config.ts +10 -3
  50. package/src/daemon.ts +613 -94
  51. package/src/dashboard/app.js +120 -0
  52. package/src/dashboard/index.html +34 -0
  53. package/src/dashboard/server.ts +267 -0
  54. package/src/dashboard/style.css +180 -0
  55. package/src/decisions.ts +39 -14
  56. package/src/diff-flags.ts +131 -241
  57. package/src/doctor.ts +795 -0
  58. package/src/escalate.ts +60 -19
  59. package/src/failure-class.ts +29 -3
  60. package/src/fleet.ts +58 -1
  61. package/src/graph-health.ts +1 -1
  62. package/src/label-projection.ts +1 -1
  63. package/src/lifecycle.ts +198 -2
  64. package/src/notices.ts +9 -0
  65. package/src/omp.ts +2 -0
  66. package/src/orchestrator-tick.ts +315 -17
  67. package/src/release-policy.ts +135 -23
  68. package/src/reports.ts +19 -5
  69. package/src/setup-host.ts +420 -8
  70. package/src/setup-install.ts +69 -14
  71. package/src/setup-wizard.ts +199 -61
  72. package/src/setup.ts +131 -35
  73. package/src/stats.ts +331 -0
  74. package/src/store.ts +206 -21
  75. package/src/tracker/github.ts +27 -4
  76. package/src/types.ts +144 -31
  77. package/src/unblock.ts +55 -11
  78. package/src/upgrade-journal.ts +220 -0
  79. package/src/upgrade-verify.ts +506 -0
  80. package/src/upgrade.ts +295 -26
  81. package/src/verbs/actions.ts +73 -1
  82. package/src/verbs/protocol.ts +29 -4
  83. package/src/verbs/server.ts +183 -20
  84. package/systemd/omp-conductor-recover.sh +433 -0
  85. package/systemd/omp-conductor.service.example +7 -0
  86. package/systemd/recover-unit-test.sh +428 -0
package/src/doctor.ts ADDED
@@ -0,0 +1,795 @@
1
+ /**
2
+ * `doctor` — mechanically check the deployment faults that have actually cost
3
+ * debugging sessions, before the fleet hits them (#287, legibility 1/5).
4
+ *
5
+ * Every production incident this reports on was environmental, silent, and
6
+ * cheap to detect: a systemd unit whose user/working-directory did not match
7
+ * the host (#152/#153 class), case-mismatched routing labels documented to
8
+ * fail silently (README Limitations), spend telemetry absent so the USD cap
9
+ * never fired, `gh` auth expiring under a live daemon. None of them had a
10
+ * check before the incident; this module is that check.
11
+ *
12
+ * Two properties are load-bearing and enforced here rather than documented:
13
+ *
14
+ * - **Read-only by default.** The store is opened as a `readonly` sqlite
15
+ * connection; nothing writes config, labels, systemd, or the database.
16
+ * The single side effect anywhere is the self-identified Telegram probe
17
+ * message, and only `--probe-telegram` enables it.
18
+ * - **Every probe is injectable.** The runner composes probes through
19
+ * {@link DoctorDeps}; the same seams the rest of the package uses — the
20
+ * tracker's own `gh` transport, `probeTelegramHealth`, the report
21
+ * outbox's send, the `renderDaemonService` runtime plan — are what a
22
+ * default run wires, so `doctor` cannot disagree with the daemon about a
23
+ * shared fact. Tests inject every probe, and no test touches the host's
24
+ * systemd unit, daemon, Telegram, GitHub account, or production state
25
+ * (the #399 boundary).
26
+ */
27
+
28
+ import { existsSync, readdirSync, readFileSync, statSync } from "node:fs";
29
+ import { spawnSync } from "node:child_process";
30
+ import { join } from "node:path";
31
+ import type { Stats } from "node:fs";
32
+ import { Database } from "bun:sqlite";
33
+
34
+ import { configBackupDir, configPath, findProject, loadConfig, resolveCaps, stateDir } from "./config.ts";
35
+ import { DEFAULT_HERDR_UNIT, probeTelegramHealth, telegramStateDir, type TelegramHealth } from "./fleet.ts";
36
+ import { planHostRuntime, STAGED_SERVICE_NAME, SYSTEMD_UNIT_DIR, totalConfiguredWorkers } from "./setup-host.ts";
37
+ import { checkTokenScopes, type ScopeCheck } from "./setup.ts";
38
+ import { dbPath, LIVE_STATES } from "./store.ts";
39
+ import { telegramReportSend, type ReportSend } from "./reports.ts";
40
+ import { fetchRateLimit, GhError, gh } from "./tracker/github.ts";
41
+ import { repoSlugFor } from "./gitops.ts";
42
+ import type { ConductorConfig, ProjectConfig, RepoTarget, RunState } from "./types.ts";
43
+
44
+ /**
45
+ * `doctor`'s finding vocabulary. `pass`/`warn`/`fail` — a warning renders the
46
+ * same but never flips the exit code; exit is nonzero iff any finding fails.
47
+ */
48
+ export type FindingStatus = "pass" | "warn" | "fail";
49
+ export type ReportStatus = "pass" | "warn" | "fail";
50
+
51
+ /** One checked fact: id (stable for CI), status, summary, and a one-line fix. */
52
+ export interface Finding {
53
+ /** Stable identifier — the JSON shape is a CI contract, so ids never move. */
54
+ id: string;
55
+ status: FindingStatus;
56
+ summary: string;
57
+ /** The one-line remedy; absent when the finding passes. */
58
+ fix?: string;
59
+ }
60
+
61
+ /**
62
+ * The stable report shape `--json` prints and the human renderer reads.
63
+ *
64
+ * `checkedAt` is the UTC ISO instant the run started; `status` is `fail` iff
65
+ * any finding is `fail`, else `warn` iff any is `warn`, else `pass`.
66
+ */
67
+ export interface DoctorReport {
68
+ project: string;
69
+ checkedAt: string;
70
+ status: ReportStatus;
71
+ findings: Finding[];
72
+ }
73
+
74
+ /** How many of the most recent completed runs are sampled for spend telemetry. */
75
+ export const SPEND_SAMPLE_RUNS = 5;
76
+
77
+ /** The past incidents this doctor flags, quoted in the finding so an operator
78
+ * who never lived them knows which failure mode the check exists to prevent. */
79
+ const INCIDENTS = {
80
+ systemd:
81
+ "the unit/user mismatch class that once killed workers at turn 0 with an empty agent.db (#152/#153)",
82
+ labels:
83
+ "label matching is exact and case-sensitive, and a case-mismatched label is documented to fail silently (README Limitations)",
84
+ spend:
85
+ "spend telemetry was once absent, so the USD cap never fired — $0.00 spend is not proof of no spend",
86
+ ghauth: "gh auth expired under a live daemon and every tracker call failed silently",
87
+ } as const;
88
+
89
+ // ------------------------------------------------------------------ dependencies
90
+
91
+ export interface GhRepoRead {
92
+ ok: boolean;
93
+ detail?: string;
94
+ }
95
+
96
+ export interface GhLabelsRead {
97
+ ok: boolean;
98
+ labels?: string[];
99
+ detail?: string;
100
+ }
101
+
102
+ /** The canonical units `setup host`/`upgrade` would stage right now. */
103
+ export interface CanonicalUnits {
104
+ daemon: string;
105
+ herdr?: string;
106
+ /** The account the daemon unit renders with, for the ownership checks. */
107
+ username: string;
108
+ }
109
+
110
+ /** One sampled run row: `state` keeps spend from in-flight rows, which are not
111
+ * a telemetry statement yet. */
112
+ export interface RunSpendRow {
113
+ state: RunState;
114
+ spendUsd: number;
115
+ }
116
+
117
+ /** Injectable seams. Every field defaults to the production wiring, which
118
+ * reuses the package's existing transports — no second GitHub client, no new
119
+ * Telegram HTTP call, no re-parsed status. */
120
+ export interface DoctorDeps {
121
+ /** Defaults to {@link loadConfig}; a loader that throws is a finding, not a crash. */
122
+ loadConfig?: () => ConductorConfig;
123
+ /** `gh auth status` scope read — the wizard's own check, so setup consent
124
+ * and `doctor` cannot disagree about what the token may do. */
125
+ scopes?: () => Promise<ScopeCheck>;
126
+ /** `gh api rate_limit` succeeds — the tracker's own liveness probe. */
127
+ rateLimitOk?: () => Promise<boolean>;
128
+ /** Whether `gh` can read one configured repo (scopes cover it). */
129
+ repoReadable?: (repo: string) => Promise<GhRepoRead>;
130
+ /** The exact labels present in one repo, for the exact-case check. */
131
+ repoLabels?: (repo: string) => Promise<GhLabelsRead>;
132
+ /** True when this is a systemd host (`/etc/systemd/system` exists). */
133
+ hasSystemd?: () => boolean;
134
+ /** The installed unit text at an absolute path, or undefined when absent. */
135
+ readUnit?: (path: string) => string | undefined;
136
+ /** stat one path; undefined when it does not exist. */
137
+ stat?: (path: string) => Stats | undefined;
138
+ /** uid of a username (`id -u`), or undefined when it cannot be resolved. */
139
+ uidOf?: (name: string) => number | undefined;
140
+ /** `PRAGMA integrity_check` over the store, read-only. ok=true for "ok". */
141
+ dbIntegrity?: (path: string) => { ok: boolean; detail?: string };
142
+ /** The most recent completed runs for one project, newest first, limited. */
143
+ recentRuns?: (project: string, limit: number) => RunSpendRow[];
144
+ /** The bot health probe `status` uses (getMe etc.). */
145
+ telegramHealth?: (projectName: string) => Promise<TelegramHealth>;
146
+ /** The report-transport send, per project; only ever invoked with
147
+ * `--probe-telegram`. Defaults to the report outbox's own transport. */
148
+ telegramSend?: (project: ProjectConfig) => ReportSend;
149
+ /** The canonical units to compare the installed units against. */
150
+ canonicalUnits?: (project: ProjectConfig, cfg: ConductorConfig) => CanonicalUnits;
151
+ /** Clock, so a run is deterministic in tests. */
152
+ now?: () => number;
153
+ /** The one opt-in side effect: send one self-identified Telegram probe. */
154
+ probeTelegram?: boolean;
155
+ }
156
+
157
+ /** The resolved probe set a run uses (defaults merged with the overrides). */
158
+ export type Probes = Required<Omit<DoctorDeps, "telegramSend" | "probeTelegram">> &
159
+ Pick<DoctorDeps, "telegramSend" | "probeTelegram">;
160
+
161
+ function messageOf(err: unknown): string {
162
+ return err instanceof Error ? err.message : String(err);
163
+ }
164
+
165
+ function firstLineOf(stderr: string): string {
166
+ const line = stderr.split("\n").find((l) => l.trim().length > 0)?.trim() ?? "";
167
+ return line.length > 120 ? `${line.slice(0, 120)}…` : line;
168
+ }
169
+
170
+ /** A gh failure → a bounded, human line (never a credential or a stack). */
171
+ function ghFailureDetail(err: unknown): string {
172
+ if (err instanceof GhError) return firstLineOf(err.stderr);
173
+ return messageOf(err);
174
+ }
175
+
176
+ // --------------------------------------------------------------- production wiring
177
+
178
+ async function defaultRepoReadable(repo: string): Promise<GhRepoRead> {
179
+ try {
180
+ await gh(["api", `repos/${repo}`]);
181
+ return { ok: true };
182
+ } catch (err) {
183
+ return { ok: false, detail: ghFailureDetail(err) };
184
+ }
185
+ }
186
+
187
+ async function defaultRepoLabels(repo: string): Promise<GhLabelsRead> {
188
+ try {
189
+ const raw = await gh(["label", "list", "--repo", repo, "--limit", "1000", "--json", "name"]);
190
+ const parsed: unknown = JSON.parse(raw);
191
+ if (!Array.isArray(parsed)) return { ok: false, detail: "gh label list returned an unexpected shape" };
192
+ const labels: string[] = [];
193
+ for (const row of parsed) {
194
+ const name = (row as { name?: unknown } | null)?.name;
195
+ if (typeof name === "string" && name.length > 0) labels.push(name);
196
+ }
197
+ return { ok: true, labels };
198
+ } catch (err) {
199
+ return { ok: false, detail: ghFailureDetail(err) };
200
+ }
201
+ }
202
+
203
+ function defaultReadUnit(path: string): string | undefined {
204
+ try {
205
+ return readFileSync(path, "utf8");
206
+ } catch {
207
+ return undefined;
208
+ }
209
+ }
210
+
211
+ function defaultStat(path: string): Stats | undefined {
212
+ try {
213
+ return statSync(path);
214
+ } catch {
215
+ return undefined;
216
+ }
217
+ }
218
+
219
+ function defaultUidOf(name: string): number | undefined {
220
+ const ran = spawnSync("id", ["-u", name], { encoding: "utf8" });
221
+ if (ran.status !== 0) return undefined;
222
+ const uid = Number.parseInt((ran.stdout ?? "").trim(), 10);
223
+ return Number.isInteger(uid) && uid >= 0 ? uid : undefined;
224
+ }
225
+
226
+ /**
227
+ * `PRAGMA integrity_check`, opened **read-only** so the check itself can never
228
+ * mutate the store it is inspecting. A missing file is a fresh host, not a
229
+ * fault; a file that is not sqlite, or a check that reports anything but
230
+ * `ok`, is.
231
+ */
232
+ function defaultDbIntegrity(path: string): { ok: boolean; detail?: string } {
233
+ if (!existsSync(path)) return { ok: true, detail: "no runs store yet — nothing to check" };
234
+ let database: Database;
235
+ try {
236
+ database = new Database(path, { readonly: true });
237
+ } catch (err) {
238
+ return { ok: false, detail: `store cannot be opened read-only: ${messageOf(err)}` };
239
+ }
240
+ try {
241
+ const rows = database.query<{ integrity_check: string }, []>("PRAGMA integrity_check").all();
242
+ for (const row of rows) {
243
+ if (row["integrity_check"] !== "ok") {
244
+ return { ok: false, detail: `integrity_check: ${row["integrity_check"]}` };
245
+ }
246
+ }
247
+ return { ok: true };
248
+ } catch (err) {
249
+ return { ok: false, detail: `integrity_check failed: ${messageOf(err)}` };
250
+ } finally {
251
+ database.close();
252
+ }
253
+ }
254
+
255
+ /**
256
+ * The most recent completed runs for one project, newest first — read on the
257
+ * same read-only connection the integrity check opens, because the store's own
258
+ * methods mutate on open and `doctor` is read-only. The state filter uses the
259
+ * store's own `LIVE_STATES` vocabulary, so `doctor` never re-decides what
260
+ * "completed" means.
261
+ */
262
+ function defaultRecentRuns(project: string, limit: number): RunSpendRow[] {
263
+ const path = dbPath();
264
+ if (!existsSync(path)) return [];
265
+ let database: Database;
266
+ try {
267
+ database = new Database(path, { readonly: true });
268
+ } catch {
269
+ return [];
270
+ }
271
+ try {
272
+ const placeholders = [...LIVE_STATES].map(() => "?").join(", ");
273
+ const rows = database
274
+ .query<{ state: string; spendUsd: number }, [string, ...string[], number]>(
275
+ `SELECT state, spendUsd FROM runs
276
+ WHERE project = ? AND state NOT IN (${placeholders})
277
+ ORDER BY startedAt DESC, rowid DESC
278
+ LIMIT ?`,
279
+ )
280
+ .all(project, ...LIVE_STATES, limit);
281
+ const out: RunSpendRow[] = [];
282
+ for (const row of rows) {
283
+ const state = row.state as RunState;
284
+ // A state vocabulary the store never had is not "completed".
285
+ if (!LIVE_STATES.includes(state)) out.push({ state, spendUsd: row.spendUsd });
286
+ }
287
+ return out;
288
+ } catch {
289
+ return [];
290
+ } finally {
291
+ database.close();
292
+ }
293
+ }
294
+
295
+ /** The canonical units — exactly what `setup host`/`upgrade` stage. */
296
+ function defaultCanonicalUnits(project: ProjectConfig, cfg: ConductorConfig): CanonicalUnits {
297
+ const caps = resolveCaps(project, cfg.defaults);
298
+ const plan = planHostRuntime(project, caps, telegramStateDir(), undefined, totalConfiguredWorkers(cfg));
299
+ const username = /^User=(.*)$/m.exec(plan.service.value)?.[1]?.trim() ?? "";
300
+ return {
301
+ daemon: plan.service.value,
302
+ ...(plan.herdrUnit === undefined ? {} : { herdr: plan.herdrUnit.value }),
303
+ username,
304
+ };
305
+ }
306
+
307
+ // --------------------------------------------------------------- finding helpers
308
+
309
+ function passFinding(id: string, summary: string): Finding {
310
+ return { id, status: "pass", summary };
311
+ }
312
+
313
+ function warnFinding(id: string, summary: string, fix: string): Finding {
314
+ return { id, status: "warn", summary, fix };
315
+ }
316
+
317
+ function failFinding(id: string, summary: string, fix: string): Finding {
318
+ return { id, status: "fail", summary, fix };
319
+ }
320
+
321
+ // ------------------------------------------------------------------ probes
322
+
323
+ function configProbe(problem: string | undefined): Finding {
324
+ if (problem === undefined) return passFinding("config", "config.json loads and validates");
325
+ return failFinding(
326
+ "config",
327
+ `config.json does not load: ${problem}`,
328
+ "repair it (run `omp-conductor setup`) before any other check can mean anything",
329
+ );
330
+ }
331
+
332
+ /**
333
+ * The config backup (#193): a pre-write copy at least as fresh as the file it
334
+ * protects. A config that was never written through the atomic writer — or
335
+ * that was hand-edited after the last save — has no backup for the bad next
336
+ * write to be rolled back from.
337
+ */
338
+ function backupProbe(probes: Probes): Finding {
339
+ const backups = (() => {
340
+ try {
341
+ return readdirSync(configBackupDir()).filter((name) => name.startsWith("config.json.bak-"));
342
+ } catch {
343
+ return [];
344
+ }
345
+ })();
346
+ const config = probes.stat(configPath());
347
+ if (config === undefined) {
348
+ return backups.length === 0
349
+ ? passFinding("config-backup", "no config.json yet — nothing to back up")
350
+ : passFinding("config-backup", `${backups.length} config backup(s) present`);
351
+ }
352
+ if (backups.length === 0) {
353
+ return warnFinding(
354
+ "config-backup",
355
+ "config.json has no backup — the next bad config write destroys the only copy",
356
+ "re-save the config (run `omp-conductor setup`) so writeConfigRaw files a pre-write backup",
357
+ );
358
+ }
359
+ const newestMtime = backups
360
+ .map((name) => probes.stat(join(configBackupDir(), name))?.mtimeMs ?? 0)
361
+ .reduce((max, m) => Math.max(max, m), 0);
362
+ if (config.mtimeMs > newestMtime + 60_000) {
363
+ return warnFinding(
364
+ "config-backup",
365
+ `config.json is newer than every backup (${backups.length} file(s)) — the newest edit was never backed up`,
366
+ "save config.json through `omp-conductor setup` so a fresh pre-write backup lands",
367
+ );
368
+ }
369
+ return passFinding("config-backup", `config backup is fresh (${backups.length} file(s))`);
370
+ }
371
+
372
+ /** The store's `PRAGMA integrity_check` — corruption was previously silent. */
373
+ function dbProbe(probes: Probes): Finding {
374
+ const result = probes.dbIntegrity(dbPath());
375
+ if (result.ok) {
376
+ return passFinding(
377
+ "db",
378
+ result.detail === undefined ? "conductor.db integrity is ok" : `conductor.db: ${result.detail}`,
379
+ );
380
+ }
381
+ return failFinding(
382
+ "db",
383
+ `conductor.db failed its integrity check: ${result.detail ?? "unknown"} — a corrupt store has been silent until nothing dispatches`,
384
+ "stop the daemon, restore the store from a backup, or rebuild it",
385
+ );
386
+ }
387
+
388
+ /** Every repo the config names: each project's tracker plus routing targets,
389
+ * as GitHub slugs, deduplicated, in config order. */
390
+ function configuredRepos(cfg: ConductorConfig | undefined): string[] {
391
+ if (cfg === undefined) return [];
392
+ const repos: string[] = [];
393
+ for (const project of cfg.projects) {
394
+ for (const repo of [
395
+ project.tracker.repo,
396
+ ...Object.values(project.routing.repos).map((r: RepoTarget) => repoSlugFor(r)),
397
+ ]) {
398
+ if (!repos.includes(repo)) repos.push(repo);
399
+ }
400
+ }
401
+ return repos;
402
+ }
403
+
404
+ /**
405
+ * gh auth: token alive, rate-limit readable, effective scopes cover every
406
+ * configured repo. The daemon's tracker reads go through the same transport,
407
+ * so a failure here is the expired-token class with the daemon already live.
408
+ */
409
+ async function ghAuthProbe(probes: Probes, repos: string[]): Promise<Finding> {
410
+ const [scopes, rateOk] = await Promise.all([probes.scopes(), probes.rateLimitOk()]);
411
+ const failures: string[] = [];
412
+ const notes: string[] = [];
413
+ if (scopes !== undefined) {
414
+ if (scopes.ok) {
415
+ notes.push(scopes.login === undefined ? "authenticated" : `authenticated as ${scopes.login}`);
416
+ notes.push(`scopes: ${scopes.scopes.join(", ") || "none listed"}`);
417
+ } else {
418
+ failures.push(
419
+ scopes.login === undefined
420
+ ? "`gh auth status` says not authenticated"
421
+ : `token for ${scopes.login} is missing scopes ${scopes.missing.join(", ") || "(none listed)"}`,
422
+ );
423
+ }
424
+ } else {
425
+ failures.push("`gh auth status` did not answer");
426
+ }
427
+ if (rateOk) notes.push("rate-limit API readable");
428
+ else failures.push("`gh api rate_limit` fails — the daemon's tracker calls would fail the same way");
429
+
430
+ const reads = await Promise.all(repos.map(async (repo) => ({ repo, read: await probes.repoReadable(repo) })));
431
+ for (const r of reads) {
432
+ if (r.read.ok) notes.push(`${r.repo} readable`);
433
+ else failures.push(`configured repo ${r.repo} is not readable with this token${r.read.detail === undefined ? "" : ` (${r.read.detail})`}`);
434
+ }
435
+ if (failures.length === 0) {
436
+ return passFinding("gh-auth", `gh is valid ${notes.length === 0 ? "" : `— ${notes.join("; ")}`}`.trim());
437
+ }
438
+ return failFinding(
439
+ "gh-auth",
440
+ `${failures.join("; ")} — ${INCIDENTS.ghauth}`,
441
+ "run `gh auth login` and grant the scopes the wizard requires (setup's checkTokenScopes names them)",
442
+ );
443
+ }
444
+
445
+ /**
446
+ * Every label the dispatch loop depends on, with exact case — a case-mismatched
447
+ * label is documented to fail silently, so doctor names the near-miss instead
448
+ * of just the absence.
449
+ */
450
+ async function labelProbe(probes: Probes, project: ProjectConfig): Promise<Finding> {
451
+ const wanted: { kind: string; name: string }[] = [
452
+ { kind: "queue", name: project.queueLabel },
453
+ { kind: "state", name: project.stateLabels.inProgress },
454
+ { kind: "state", name: project.stateLabels.blocked },
455
+ { kind: "state", name: project.stateLabels.failed },
456
+ ...Object.keys(project.routing.repos).map((key) => ({
457
+ kind: "routing",
458
+ name: `${project.routing.labelPrefix}${key}`,
459
+ })),
460
+ ];
461
+ const read = await probes.repoLabels(project.tracker.repo);
462
+ if (!read.ok) {
463
+ return failFinding(
464
+ "labels",
465
+ `cannot read the labels of ${project.tracker.repo}${read.detail === undefined ? "" : ` (${read.detail})`}`,
466
+ "check `gh` can list that repo's labels — the daemon reads them the same way",
467
+ );
468
+ }
469
+ const present = new Set(read.labels ?? []);
470
+ const norm = (s: string): string => s.toLowerCase().replace(/[-_\s:]+/g, "");
471
+ const nearByName = new Map<string, string>();
472
+ for (const label of read.labels ?? []) {
473
+ if (!nearByName.has(norm(label))) nearByName.set(norm(label), label);
474
+ }
475
+ const missing = wanted.filter((w) => !present.has(w.name));
476
+ if (missing.length === 0) {
477
+ return passFinding("labels", `${wanted.length} configured label(s) on ${project.tracker.repo} match exactly`);
478
+ }
479
+ const detail = missing.map((m) => {
480
+ const near = nearByName.get(norm(m.name));
481
+ return near === undefined || near === m.name ? `"${m.name}" (${m.kind})` : `"${m.name}" (${m.kind}) — near-miss present: "${near}"`;
482
+ });
483
+ const allNearMisses = missing.every((m) => {
484
+ const near = nearByName.get(norm(m.name));
485
+ return near !== undefined && near !== m.name;
486
+ });
487
+ return failFinding(
488
+ "labels",
489
+ `${detail.join("; ")} — ${INCIDENTS.labels}`,
490
+ allNearMisses
491
+ ? "fix the label case/format in the tracker repo (or change the config to the actual spelling)"
492
+ : "create the missing label(s) in the tracker repo (`gh label create`, or re-run `omp-conductor setup`)",
493
+ );
494
+ }
495
+
496
+ /** Line-level drift between an installed unit and the canonical render. */
497
+ export function unitDrift(installed: string, canonical: string): { differences: string[] } {
498
+ const want = canonical.split("\n").map((l) => l.trim()).filter((l) => l.length > 0);
499
+ const have = installed.split("\n").map((l) => l.trim()).filter((l) => l.length > 0);
500
+ const wantSet = new Set(want);
501
+ const haveSet = new Set(have);
502
+ const differences = [
503
+ ...want.filter((l) => !haveSet.has(l)).map((l) => `missing/differs: ${l}`),
504
+ ...have.filter((l) => !wantSet.has(l)).map((l) => `extra: ${l}`),
505
+ ];
506
+ return { differences };
507
+ }
508
+
509
+ function driftLines(name: string, installed: string, canonical: string): string[] {
510
+ const { differences } = unitDrift(installed, canonical);
511
+ if (differences.length === 0) return [];
512
+ const shown = differences.slice(0, 4).join("; ");
513
+ const more = differences.length > 4 ? ` (+${differences.length - 4} more)` : "";
514
+ return [`${name} differs from the staged unit: ${shown}${more}`];
515
+ }
516
+
517
+ /**
518
+ * Installed unit vs the canonical staged rendering: any drift in User=,
519
+ * Environment, WorkingDirectory, ExecStart, Restart/SuccessExitStatus, or
520
+ * memory lines is the unit-drift class that once killed workers at turn 0.
521
+ */
522
+ function unitProbe(probes: Probes, project: ProjectConfig | undefined, cfg: ConductorConfig | undefined): Finding {
523
+ if (!probes.hasSystemd()) return passFinding("systemd-unit", "no systemd directory on this host — nothing to check");
524
+ if (project === undefined || cfg === undefined) {
525
+ return passFinding("systemd-unit", "config unreadable — the canonical staged unit is unknown, nothing to compare");
526
+ }
527
+ const canonical = probes.canonicalUnits(project, cfg);
528
+ const installedDaemon = probes.readUnit(join(SYSTEMD_UNIT_DIR, STAGED_SERVICE_NAME));
529
+ if (installedDaemon === undefined) {
530
+ return warnFinding("systemd-unit", `${STAGED_SERVICE_NAME} is not installed`, "install it: run `omp-conductor setup host` from the fleet account");
531
+ }
532
+ const problems = driftLines(STAGED_SERVICE_NAME, installedDaemon, canonical.daemon);
533
+ if (canonical.herdr !== undefined) {
534
+ const installedHerdr = probes.readUnit(join(SYSTEMD_UNIT_DIR, DEFAULT_HERDR_UNIT));
535
+ if (installedHerdr === undefined) {
536
+ problems.push(`${DEFAULT_HERDR_UNIT} is not installed (the staged plan provisions it)`);
537
+ } else {
538
+ problems.push(...driftLines(DEFAULT_HERDR_UNIT, installedHerdr, canonical.herdr));
539
+ }
540
+ }
541
+ if (problems.length === 0) {
542
+ return passFinding("systemd-unit", "installed units match the staged render");
543
+ }
544
+ return failFinding(
545
+ "systemd-unit",
546
+ [...problems, INCIDENTS.systemd].join("; "),
547
+ "re-stage and reinstall with `omp-conductor setup host` from the fleet account",
548
+ );
549
+ }
550
+
551
+ function installedUnitUser(probes: Probes): string | undefined {
552
+ const unit = probes.readUnit(join(SYSTEMD_UNIT_DIR, STAGED_SERVICE_NAME));
553
+ if (unit === undefined) return undefined;
554
+ const match = /^User=(.*)$/m.exec(unit);
555
+ if (match === null) return undefined;
556
+ const value = match[1]?.trim().replace(/^"(.*)"$/, "$1") ?? "";
557
+ return value === "" ? undefined : value;
558
+ }
559
+
560
+ /**
561
+ * Ownership and modes of the unit file and the runtime dirs: a daemon that
562
+ * cannot write its own state is the empty-agent.db class, and a unit someone
563
+ * else can edit is a takeover.
564
+ */
565
+ function ownershipProbe(probes: Probes): Finding {
566
+ if (!probes.hasSystemd()) return passFinding("systemd-ownership", "no systemd directory on this host — nothing to check");
567
+ const failures: string[] = [];
568
+ const warns: string[] = [];
569
+ const daemonPath = join(SYSTEMD_UNIT_DIR, STAGED_SERVICE_NAME);
570
+ const daemonStat = probes.stat(daemonPath);
571
+ if (daemonStat !== undefined) {
572
+ if (daemonStat.uid !== 0) failures.push(`${STAGED_SERVICE_NAME} is not owned by root`);
573
+ if ((daemonStat.mode & 0o022) !== 0) warns.push(`${STAGED_SERVICE_NAME} is group/other-writable`);
574
+ }
575
+ const unitUser = installedUnitUser(probes);
576
+ const fleetUid = unitUser === undefined ? process.getuid?.() : probes.uidOf(unitUser);
577
+ for (const [label, dir] of [["state dir", stateDir()], ["telegram state dir", telegramStateDir()]] as const) {
578
+ const st = probes.stat(dir);
579
+ if (st === undefined) continue;
580
+ if (fleetUid !== undefined && st.uid !== fleetUid) {
581
+ failures.push(
582
+ `${label} (${dir}) is owned by uid ${st.uid}, but the ${unitUser === undefined ? "current user" : `unit's account "${unitUser}"`} has uid ${fleetUid}`,
583
+ );
584
+ }
585
+ if ((st.mode & 0o022) !== 0) warns.push(`${label} (${dir}) is group/other-writable`);
586
+ }
587
+ if (failures.length === 0 && warns.length === 0) {
588
+ return passFinding("systemd-ownership", "unit file and runtime dirs are owned and permitted as deployed");
589
+ }
590
+ const summary = [...failures, ...warns].join("; ");
591
+ const fix = "reinstall the unit as root and chown the state dirs to the account the unit runs as";
592
+ if (failures.length === 0) {
593
+ return warnFinding("systemd-ownership", summary, fix);
594
+ }
595
+ return failFinding("systemd-ownership", summary, fix);
596
+ }
597
+
598
+ /** IANA timezone in reporting config — an invalid zone silently mis-schedules
599
+ * the availability window and the daily digest (#273). */
600
+ function timezoneProbe(project: ProjectConfig | undefined): Finding {
601
+ if (project === undefined) return passFinding("reporting-timezone", "no project resolved — nothing to check");
602
+ const tzs: [string, string | undefined][] = [
603
+ ["reporting.availability.timezone", project.reporting?.availability?.timezone],
604
+ ["reporting.digest.timezone", project.reporting?.digest?.timezone],
605
+ ];
606
+ const configured = tzs.filter(([, tz]) => tz !== undefined);
607
+ const bad = configured.filter(([, tz]) => tz !== undefined && !isKnownTimezone(tz));
608
+ if (bad.length === 0) {
609
+ return passFinding(
610
+ "reporting-timezone",
611
+ configured.length === 0
612
+ ? "no reporting timezone configured"
613
+ : `reporting timezone(s) valid: ${configured.map(([, tz]) => `"${tz}"`).join(", ")}`,
614
+ );
615
+ }
616
+ return failFinding(
617
+ "reporting-timezone",
618
+ `${bad.map(([label, tz]) => `${label}: "${tz}" is not a known IANA timezone`).join("; ")} — an invalid zone silently skips the availability window and the daily digest`,
619
+ "set a valid IANA timezone (e.g. Europe/London) in reporting.availability or reporting.digest",
620
+ );
621
+ }
622
+
623
+ /** The same acceptance the config validator uses (`Intl.DateTimeFormat`), so
624
+ * `doctor` and `config.ts` cannot disagree about a zone. */
625
+ export function isKnownTimezone(tz: string): boolean {
626
+ try {
627
+ new Intl.DateTimeFormat("en-GB", { timeZone: tz });
628
+ return true;
629
+ } catch {
630
+ return false;
631
+ }
632
+ }
633
+
634
+ /** The self-identified probe message `--probe-telegram` delivers through the
635
+ * same report transport a real report would take. */
636
+ function probeMessage(project: string, checkedAt: string): string {
637
+ return [
638
+ "omp-conductor doctor · Telegram delivery probe",
639
+ `project: ${project} · started ${checkedAt}`,
640
+ "Sent by `omp-conductor doctor --probe-telegram` to verify delivery end to end.",
641
+ "No action needed — you can ignore this message.",
642
+ ].join("\n");
643
+ }
644
+
645
+ /**
646
+ * Telegram delivery health — and, when `--probe-telegram` is set, one
647
+ * self-identified message through the report transport. A dead bot token is
648
+ * silent: reports leave the outbox and never land.
649
+ */
650
+ async function telegramProbe(probes: Probes, project: ProjectConfig | undefined, checkedAt: number): Promise<Finding> {
651
+ if (project === undefined) return passFinding("telegram", "no project resolved — nothing to check");
652
+ const chatId = (project.escalation.telegramChatId ?? "").trim();
653
+ if (chatId === "") {
654
+ return passFinding("telegram", "no escalation.telegramChatId configured — nothing to probe");
655
+ }
656
+ const health = await probes.telegramHealth(project.name);
657
+ const failures: string[] = [];
658
+ const notes: string[] = [];
659
+ if (health.kind === "down") failures.push(`bot health: down (${health.detail ?? "getMe failed"})`);
660
+ else if (health.kind === "unconfigured") notes.push(`bot health: unconfigured (${health.detail ?? "no token"})`);
661
+ else if (health.kind === "degraded") notes.push(`bot health: degraded (${health.detail ?? ""})`);
662
+ else notes.push(`bot health: ok (${health.detail ?? ""})`);
663
+ if (probes.probeTelegram) {
664
+ const send = probes.telegramSend?.(project);
665
+ if (send === undefined) {
666
+ failures.push("probe requested but no send transport is wired");
667
+ } else {
668
+ try {
669
+ await send(probeMessage(project.name, new Date(checkedAt).toISOString()));
670
+ notes.push("probe message delivered");
671
+ } catch (err) {
672
+ failures.push(`probe send failed: ${messageOf(err)}`);
673
+ }
674
+ }
675
+ }
676
+ if (failures.length === 0) {
677
+ return passFinding("telegram", notes.join("; "));
678
+ }
679
+ return failFinding(
680
+ "telegram",
681
+ failures.join("; "),
682
+ "fix the bot token / omp-telegram install; `doctor --probe-telegram` proves delivery end to end",
683
+ );
684
+ }
685
+
686
+ /** Spend telemetry: warn when the last K completed runs all recorded $0.00 —
687
+ * the USD cap cannot fire on zeros ($0.00 is not proof of no spend). */
688
+ function spendProbe(rows: RunSpendRow[], limit: number): Finding {
689
+ if (rows.length < limit) {
690
+ return passFinding("spend-telemetry", `${rows.length} completed run(s) observed — fewer than ${limit}, nothing to judge yet`);
691
+ }
692
+ const window = rows.slice(0, limit);
693
+ if (window.every((r) => r.spendUsd === 0)) {
694
+ return warnFinding(
695
+ "spend-telemetry",
696
+ `the last ${limit} completed runs all recorded $0.00 spend — ${INCIDENTS.spend}`,
697
+ "verify harness spend reporting (the per-run spendUsd column / `omp usage --json`); a USD cap on top of zeros never fires",
698
+ );
699
+ }
700
+ const total = window.reduce((sum, r) => sum + r.spendUsd, 0);
701
+ return passFinding("spend-telemetry", `spend observed on the last ${window.length} completed runs ($${total.toFixed(2)} total)`);
702
+ }
703
+
704
+ // ------------------------------------------------------------- composition
705
+
706
+ /**
707
+ * Run every probe and assemble the stable report.
708
+ *
709
+ * @param projectName the `--project NAME` value (undefined on single-project hosts)
710
+ * @param opts injected seams; every probe defaults to the production wiring
711
+ */
712
+ export async function runDoctor(projectName: string | undefined, opts: DoctorDeps = {}): Promise<DoctorReport> {
713
+ const probes: Probes = { ...defaultProbes(), ...opts };
714
+ const checkedAt = probes.now();
715
+ const checkedAtIso = new Date(checkedAt).toISOString();
716
+
717
+ let cfg: ConductorConfig | undefined;
718
+ let configProblem: string | undefined;
719
+ try {
720
+ cfg = probes.loadConfig();
721
+ } catch (err) {
722
+ configProblem = messageOf(err);
723
+ }
724
+
725
+ let project: ProjectConfig | undefined;
726
+ let projectProblem: string | undefined;
727
+ if (cfg !== undefined) {
728
+ try {
729
+ project = findProject(cfg, projectName);
730
+ } catch (err) {
731
+ projectProblem = messageOf(err);
732
+ }
733
+ }
734
+
735
+ const findings: Finding[] = [];
736
+ findings.push(configProbe(configProblem));
737
+ findings.push(
738
+ cfg === undefined
739
+ ? passFinding("project", "config unreadable — no project to resolve")
740
+ : project === undefined
741
+ ? failFinding(
742
+ "project",
743
+ `no project resolved: ${projectProblem ?? "unknown error"}`,
744
+ "pass --project NAME (doctor checks one project per run)",
745
+ )
746
+ : passFinding("project", project.name),
747
+ );
748
+ findings.push(backupProbe(probes));
749
+ findings.push(dbProbe(probes));
750
+ findings.push(await ghAuthProbe(probes, configuredRepos(cfg)));
751
+ findings.push(
752
+ project === undefined
753
+ ? passFinding("labels", projectProblem === undefined ? "no project resolved — nothing to check" : `labels uncheckable: ${projectProblem}`)
754
+ : await labelProbe(probes, project),
755
+ );
756
+ findings.push(unitProbe(probes, project, cfg));
757
+ findings.push(ownershipProbe(probes));
758
+ findings.push(timezoneProbe(project));
759
+ findings.push(await telegramProbe(probes, project, checkedAt));
760
+ findings.push(spendProbe(project === undefined ? [] : probes.recentRuns(project.name, SPEND_SAMPLE_RUNS), SPEND_SAMPLE_RUNS));
761
+
762
+ const status: ReportStatus = findings.some((f) => f.status === "fail")
763
+ ? "fail"
764
+ : findings.some((f) => f.status === "warn")
765
+ ? "warn"
766
+ : "pass";
767
+ return {
768
+ project: project === undefined ? projectName ?? "(none resolved)" : project.name,
769
+ checkedAt: checkedAtIso,
770
+ status,
771
+ findings,
772
+ };
773
+ }
774
+
775
+ /** The default wiring — every production transport the rest of the package uses. */
776
+ export function defaultProbes(): Probes {
777
+ return {
778
+ loadConfig,
779
+ scopes: checkTokenScopes,
780
+ rateLimitOk: async () => (await fetchRateLimit()) !== undefined,
781
+ repoReadable: defaultRepoReadable,
782
+ repoLabels: defaultRepoLabels,
783
+ hasSystemd: () => existsSync(SYSTEMD_UNIT_DIR),
784
+ readUnit: defaultReadUnit,
785
+ stat: defaultStat,
786
+ uidOf: defaultUidOf,
787
+ dbIntegrity: defaultDbIntegrity,
788
+ recentRuns: defaultRecentRuns,
789
+ telegramHealth: (projectName) => probeTelegramHealth(projectName),
790
+ telegramSend: telegramReportSend,
791
+ canonicalUnits: defaultCanonicalUnits,
792
+ now: Date.now,
793
+ probeTelegram: false,
794
+ };
795
+ }