@celilo/cli 1.11.0 → 1.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CELILO_CORE_MODULES.md +2 -1
- package/CELILO_SUBSYSTEMS.md +17 -2
- package/package.json +3 -3
- package/src/cli/commands/alerts-list.ts +16 -1
- package/src/cli/commands/backup-list.test.ts +82 -1
- package/src/cli/commands/backup-list.ts +113 -4
- package/src/cli/commands/console.ts +122 -0
- package/src/cli/commands/module-list.ts +3 -41
- package/src/cli/commands/module-publish.ts +2 -0
- package/src/cli/completion.ts +5 -0
- package/src/cli/index.ts +25 -1
- package/src/console/closure.test.ts +246 -0
- package/src/console/closure.ts +208 -0
- package/src/console/control-plane-boundary.test.ts +75 -0
- package/src/console/projection.test.ts +231 -0
- package/src/console/projection.ts +327 -0
- package/src/db/schema.ts +19 -14
- package/src/hooks/broker.test.ts +4 -6
- package/src/hooks/executor.test.ts +85 -4
- package/src/hooks/executor.ts +164 -9
- package/src/hooks/hook-jail-unreachability.test.ts +173 -0
- package/src/hooks/hook-state-dir.test.ts +14 -2
- package/src/hooks/hook-timeout.test.ts +2 -4
- package/src/hooks/hook-trespass.test.ts +50 -5
- package/src/hooks/jail.test.ts +370 -0
- package/src/hooks/jail.ts +491 -0
- package/src/hooks/mount-set.ts +24 -0
- package/src/hooks/test-fixtures/jail-probe-hook.ts +59 -0
- package/src/manifest/icon-schema.test.ts +48 -0
- package/src/manifest/schema.ts +92 -0
- package/src/manifest/validate.test.ts +142 -0
- package/src/manifest/validate.ts +101 -0
- package/src/module/import.test.ts +116 -0
- package/src/module/import.ts +73 -1
- package/src/module/packaging/audit.ts +103 -1
- package/src/module/packaging/classify-module-path.test.ts +36 -0
- package/src/module/packaging/package-rules.ts +18 -0
- package/src/policy/capability-shape-baseline.ts +8 -0
- package/src/policy/module-business-baseline.ts +12 -0
- package/src/registry/client.ts +9 -0
- package/src/services/alerting/observed-health.ts +71 -0
- package/src/services/api-principal-enrolment.test.ts +179 -0
- package/src/services/api-principal-enrolment.ts +103 -0
- package/src/services/audit/backups.ts +10 -1
- package/src/services/backup-metadata.ts +19 -11
- package/src/services/consumer-cleanup.ts +31 -5
- package/src/services/instance-ops.test.ts +302 -0
- package/src/services/instance-ops.ts +292 -0
- package/src/services/module-instances.test.ts +428 -42
- package/src/services/module-instances.ts +219 -26
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The console projection, against a real schema.
|
|
3
|
+
*
|
|
4
|
+
* The payload SHAPE is the point of these reads, so the assertions are about
|
|
5
|
+
* what is present and what is deliberately absent, not only about values.
|
|
6
|
+
*/
|
|
7
|
+
import { afterEach, beforeEach, describe, expect, test } from 'bun:test';
|
|
8
|
+
import type { DbClient } from '../db/client';
|
|
9
|
+
import { NETWORK_ZONES } from '../db/schema';
|
|
10
|
+
import { cleanupTestDatabase, setupTestDatabase } from '../test-utils/database';
|
|
11
|
+
import { consoleStatus } from './projection';
|
|
12
|
+
|
|
13
|
+
/** A manifest big enough to notice if it ever leaked into the payload. */
|
|
14
|
+
const FAT_MANIFEST = JSON.stringify({
|
|
15
|
+
id: 'caddy-internal',
|
|
16
|
+
padding: 'x'.repeat(4096),
|
|
17
|
+
requires: { capabilities: [{ name: 'idp' }] },
|
|
18
|
+
});
|
|
19
|
+
|
|
20
|
+
describe('consoleStatus', () => {
|
|
21
|
+
let db: DbClient;
|
|
22
|
+
|
|
23
|
+
beforeEach(async () => {
|
|
24
|
+
db = await setupTestDatabase();
|
|
25
|
+
db.$client.run(
|
|
26
|
+
`INSERT INTO modules (id, name, version, source_path, manifest_data, state)
|
|
27
|
+
VALUES ('caddy-internal', 'Caddy', '1.2.0', '/path', '${FAT_MANIFEST}', 'VERIFIED')`,
|
|
28
|
+
);
|
|
29
|
+
db.$client.run(
|
|
30
|
+
`INSERT INTO modules (id, name, version, source_path, manifest_data, state)
|
|
31
|
+
VALUES ('api-only', 'API only', '0.1.0', '/path', '{}', 'VERIFIED')`,
|
|
32
|
+
);
|
|
33
|
+
db.$client.run(
|
|
34
|
+
`INSERT INTO module_systems (module_id, name, hostname, ipv4_address, zone, infra_type)
|
|
35
|
+
VALUES ('caddy-internal', 'web', 'caddy-int', '10.0.10.14', 'dmz', 'container_service')`,
|
|
36
|
+
);
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
afterEach(async () => {
|
|
40
|
+
await cleanupTestDatabase(db);
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
test('serves the canonical zone order rather than a copy', () => {
|
|
44
|
+
// A zone added to NETWORK_ZONES must reach the console without a console
|
|
45
|
+
// release, so the payload carries the list itself.
|
|
46
|
+
expect(consoleStatus(db).zones).toEqual([...NETWORK_ZONES]);
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
test('does NOT carry manifestData', () => {
|
|
50
|
+
// The whole reason these reads exist. `module list --json` is 156 KB for 23
|
|
51
|
+
// modules because it embeds this blob on every row.
|
|
52
|
+
const payload = JSON.stringify(consoleStatus(db));
|
|
53
|
+
expect(payload).not.toContain('padding');
|
|
54
|
+
expect(payload.length).toBeLessThan(2000);
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
test('carries a module with no system rather than omitting it', () => {
|
|
58
|
+
// A closure that names a module the topology then cannot draw is
|
|
59
|
+
// indistinguishable from a defect, so the payload keeps it with an empty
|
|
60
|
+
// systems list and the renderer decides where to put it.
|
|
61
|
+
const found = consoleStatus(db).modules.find((m) => m.id === 'api-only');
|
|
62
|
+
expect(found).toBeDefined();
|
|
63
|
+
expect(found?.systems).toEqual([]);
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
test('carries each system with the address recorded in deployment state', () => {
|
|
67
|
+
const caddy = consoleStatus(db).modules.find((m) => m.id === 'caddy-internal');
|
|
68
|
+
expect(caddy?.systems).toEqual([{ hostname: 'caddy-int', address: '10.0.10.14', zone: 'dmz' }]);
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
test('a deployed module with no monitor reads as not observed, not as healthy', () => {
|
|
72
|
+
const caddy = consoleStatus(db).modules.find((m) => m.id === 'caddy-internal');
|
|
73
|
+
expect(caddy?.health.cell).toBe('not observed');
|
|
74
|
+
expect(caddy?.health.monitored).toBe(false);
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
test('an undeployed module is distinguished from an unobserved one', () => {
|
|
78
|
+
db.$client.run(
|
|
79
|
+
`INSERT INTO modules (id, name, version, source_path, manifest_data, state)
|
|
80
|
+
VALUES ('imported', 'Imported', '0.1.0', '/path', '{}', 'IMPORTED')`,
|
|
81
|
+
);
|
|
82
|
+
const imported = consoleStatus(db).modules.find((m) => m.id === 'imported');
|
|
83
|
+
// "not observed" is a finding about a deployed module. Saying it about a
|
|
84
|
+
// module that was never deployed would be noise.
|
|
85
|
+
expect(imported?.health.cell).toBe('not deployed');
|
|
86
|
+
});
|
|
87
|
+
|
|
88
|
+
test('reports no backup as null rather than as zero', () => {
|
|
89
|
+
const caddy = consoleStatus(db).modules.find((m) => m.id === 'caddy-internal');
|
|
90
|
+
expect(caddy?.lastBackupAt).toBeNull();
|
|
91
|
+
expect(caddy?.lastBackupFailed).toBe(false);
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
test('modules come back in a stable order', () => {
|
|
95
|
+
const first = consoleStatus(db).modules.map((m) => m.id);
|
|
96
|
+
const second = consoleStatus(db).modules.map((m) => m.id);
|
|
97
|
+
expect(first).toEqual(second);
|
|
98
|
+
expect(first).toEqual([...first].sort());
|
|
99
|
+
});
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* The three facts the roster's lamps read, and the unit the ages are in.
|
|
104
|
+
*
|
|
105
|
+
* Every one of these is a wrong-but-plausible number rather than a crash. A
|
|
106
|
+
* timestamp in the wrong unit renders as an age that looks like a fresh backup.
|
|
107
|
+
* A module reported as watched when nothing runs its monitor reads as healthy.
|
|
108
|
+
* Neither throws, and neither is visible in review.
|
|
109
|
+
*/
|
|
110
|
+
describe('consoleStatus derived facts', () => {
|
|
111
|
+
let db: DbClient;
|
|
112
|
+
|
|
113
|
+
const DAY = 86_400_000;
|
|
114
|
+
const HOOKED = JSON.stringify({
|
|
115
|
+
id: 'forgejo',
|
|
116
|
+
hooks: { on_backup: { script: './backup.ts' } },
|
|
117
|
+
backup: { schedule: 'daily' },
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
function moduleRow(id: string, manifest: string): void {
|
|
121
|
+
db.$client.run(
|
|
122
|
+
`INSERT INTO modules (id, name, version, source_path, manifest_data, state)
|
|
123
|
+
VALUES ('${id}', '${id}', '1.0.0', '/path', '${manifest}', 'VERIFIED')`,
|
|
124
|
+
);
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function backupRow(id: string, moduleId: string, status: string, completedAtMs: number): void {
|
|
128
|
+
db.$client.run(
|
|
129
|
+
`INSERT INTO backup_storages (id, storage_id, name, provider_name, credentials_encrypted, provider_config)
|
|
130
|
+
VALUES ('st-1', 'aws', 'AWS', 's3', 'x', '{}') ON CONFLICT DO NOTHING`,
|
|
131
|
+
);
|
|
132
|
+
db.$client.run(
|
|
133
|
+
`INSERT INTO backups (id, module_id, storage_id, storage_path, backup_type, status, started_at, completed_at, metadata)
|
|
134
|
+
VALUES ('${id}', '${moduleId}', 'st-1', 'p/${id}', 'module_data', '${status}',
|
|
135
|
+
${Math.floor(completedAtMs / 1000)}, ${Math.floor(completedAtMs / 1000)}, '{}')`,
|
|
136
|
+
);
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
function monitorRow(target: string, opts: { policy?: boolean; ran?: boolean }): void {
|
|
140
|
+
db.$client.run(
|
|
141
|
+
`INSERT INTO monitors (id, kind, target, interval_minutes, enabled, escalation_policy_id, last_run_at)
|
|
142
|
+
VALUES ('mon-${target}', 'module_hook', '${target}', 60, 1, ${opts.policy ? "'pol-1'" : 'NULL'},
|
|
143
|
+
${opts.ran ? Math.floor(Date.now() / 1000) : 'NULL'})`,
|
|
144
|
+
);
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
function moduleOf(id: string) {
|
|
148
|
+
const found = consoleStatus(db).modules.find((m) => m.id === id);
|
|
149
|
+
if (!found) throw new Error(`${id} missing from the projection`);
|
|
150
|
+
return found;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
beforeEach(async () => {
|
|
154
|
+
db = await setupTestDatabase();
|
|
155
|
+
db.$client.run(
|
|
156
|
+
`INSERT INTO escalation_policies (id, name) VALUES ('pol-1', 'oncall') ON CONFLICT DO NOTHING`,
|
|
157
|
+
);
|
|
158
|
+
});
|
|
159
|
+
|
|
160
|
+
afterEach(async () => {
|
|
161
|
+
await cleanupTestDatabase(db);
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
test('lastBackupAt is epoch MILLISECONDS', () => {
|
|
165
|
+
// The whole reason this test exists. It was seconds, the protocol on the
|
|
166
|
+
// other side says milliseconds, and nothing joined the two yet — so the
|
|
167
|
+
// factor of a thousand sat there looking like working code.
|
|
168
|
+
const at = Date.now() - 2 * DAY;
|
|
169
|
+
moduleRow('forgejo', HOOKED);
|
|
170
|
+
backupRow('b1', 'forgejo', 'completed', at);
|
|
171
|
+
|
|
172
|
+
const lastBackupAt = moduleOf('forgejo').lastBackupAt;
|
|
173
|
+
expect(lastBackupAt).not.toBeNull();
|
|
174
|
+
// Within a day of now in ms. In seconds this would be ~1.8 billion, which
|
|
175
|
+
// is fifty-odd years, and would render as an absurd age rather than throw.
|
|
176
|
+
expect(Date.now() - (lastBackupAt as number)).toBeLessThan(3 * DAY);
|
|
177
|
+
});
|
|
178
|
+
|
|
179
|
+
test('a module with no monitor at all is unwatched', () => {
|
|
180
|
+
moduleRow('forgejo', HOOKED);
|
|
181
|
+
expect(moduleOf('forgejo').unwatched).toBe(true);
|
|
182
|
+
});
|
|
183
|
+
|
|
184
|
+
test('a monitor that has NEVER RUN is not coverage', () => {
|
|
185
|
+
// It produces exactly as much evidence as no monitor does.
|
|
186
|
+
moduleRow('forgejo', HOOKED);
|
|
187
|
+
monitorRow('forgejo', { policy: true, ran: false });
|
|
188
|
+
expect(moduleOf('forgejo').unwatched).toBe(true);
|
|
189
|
+
});
|
|
190
|
+
|
|
191
|
+
test('a monitor that runs makes the module watched', () => {
|
|
192
|
+
moduleRow('forgejo', HOOKED);
|
|
193
|
+
monitorRow('forgejo', { policy: true, ran: true });
|
|
194
|
+
expect(moduleOf('forgejo').unwatched).toBe(false);
|
|
195
|
+
});
|
|
196
|
+
|
|
197
|
+
test('a running monitor with no escalation policy pages nobody', () => {
|
|
198
|
+
// Worse than unwatched in one respect: it looks monitored. The alert is
|
|
199
|
+
// raised on time and reaches no one.
|
|
200
|
+
moduleRow('forgejo', HOOKED);
|
|
201
|
+
monitorRow('forgejo', { policy: false, ran: true });
|
|
202
|
+
const module = moduleOf('forgejo');
|
|
203
|
+
expect(module.unwatched).toBe(false);
|
|
204
|
+
expect(module.pagesNobody).toBe(true);
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
test('an overdue daily backup is stale', () => {
|
|
208
|
+
moduleRow('forgejo', HOOKED);
|
|
209
|
+
backupRow('b1', 'forgejo', 'completed', Date.now() - 5 * DAY);
|
|
210
|
+
expect(moduleOf('forgejo').backupStale).toBe(true);
|
|
211
|
+
});
|
|
212
|
+
|
|
213
|
+
test('a fresh daily backup is not', () => {
|
|
214
|
+
moduleRow('forgejo', HOOKED);
|
|
215
|
+
backupRow('b1', 'forgejo', 'completed', Date.now() - 60_000);
|
|
216
|
+
expect(moduleOf('forgejo').backupStale).toBe(false);
|
|
217
|
+
});
|
|
218
|
+
|
|
219
|
+
test('a module with NO on_backup hook is never stale', () => {
|
|
220
|
+
// There is nothing to run. Reporting eighteen such modules as overdue is
|
|
221
|
+
// how this page becomes an alarm nobody can act on (celilo#1131).
|
|
222
|
+
moduleRow('caddy', JSON.stringify({ id: 'caddy' }));
|
|
223
|
+
expect(moduleOf('caddy').backupStale).toBe(false);
|
|
224
|
+
});
|
|
225
|
+
|
|
226
|
+
test('a hooked module that has NEVER been backed up is stale', () => {
|
|
227
|
+
// Distinct from the case above: this one was supposed to run and did not.
|
|
228
|
+
moduleRow('forgejo', HOOKED);
|
|
229
|
+
expect(moduleOf('forgejo').backupStale).toBe(true);
|
|
230
|
+
});
|
|
231
|
+
});
|
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The narrow reads the web console polls.
|
|
3
|
+
*
|
|
4
|
+
* Separate from the CLI's own reads for a measured reason: `celilo module list
|
|
5
|
+
* --json` returns 156 KB for 23 modules because it embeds every module's
|
|
6
|
+
* `manifestData` blob. That is the right payload for a human debugging one
|
|
7
|
+
* module and the wrong one for a loop that runs every few seconds, so the
|
|
8
|
+
* console gets a projection with the manifest left out and the two derived
|
|
9
|
+
* facts it actually renders (observed health, backup freshness) already
|
|
10
|
+
* computed.
|
|
11
|
+
*
|
|
12
|
+
* Reads only. Nothing here writes, and the console's API principal is granted
|
|
13
|
+
* only read ops, so a write added here would fail at the boundary rather than
|
|
14
|
+
* succeed quietly.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { desc, eq } from 'drizzle-orm';
|
|
18
|
+
import { z } from 'zod';
|
|
19
|
+
import type { DbClient } from '../db/client';
|
|
20
|
+
import { NETWORK_ZONES, backups, moduleSystems, modules, monitors } from '../db/schema';
|
|
21
|
+
import { loadObservedHealthDetail } from '../services/alerting/observed-health';
|
|
22
|
+
import { loadBackupAuditInfo } from '../services/audit/backup-source';
|
|
23
|
+
import { backupStaleThresholdMs, moduleHasBackupHook } from '../services/audit/backups';
|
|
24
|
+
import { effectiveBackupSchedule } from '../services/backup-schedule';
|
|
25
|
+
import { listCapabilityBindings } from '../services/capability-bindings';
|
|
26
|
+
import type { ConsumedCapabilities } from '../services/consumer-cleanup';
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Just the capability lists, validated at the boundary.
|
|
30
|
+
*
|
|
31
|
+
* `modules.manifestData` is opaque JSON in the database, which is a trust
|
|
32
|
+
* boundary even though celilo wrote it: rows outlive the code that wrote them,
|
|
33
|
+
* and an older celilo's row is exactly the case a clean-database test can never
|
|
34
|
+
* produce. `.catch` rather than `.parse` because one unreadable manifest must
|
|
35
|
+
* end that branch of the walk, not blank the whole topology.
|
|
36
|
+
*/
|
|
37
|
+
const CONSUMED_CAPABILITIES = z
|
|
38
|
+
.object({
|
|
39
|
+
requires: z
|
|
40
|
+
.object({ capabilities: z.array(z.object({ name: z.string() })).optional() })
|
|
41
|
+
.optional(),
|
|
42
|
+
optional: z
|
|
43
|
+
.object({ capabilities: z.array(z.object({ name: z.string() })).optional() })
|
|
44
|
+
.optional(),
|
|
45
|
+
})
|
|
46
|
+
.catch({});
|
|
47
|
+
|
|
48
|
+
export interface ConsoleSystem {
|
|
49
|
+
hostname: string;
|
|
50
|
+
address: string;
|
|
51
|
+
zone: string;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export interface ConsoleModule {
|
|
55
|
+
id: string;
|
|
56
|
+
version: string;
|
|
57
|
+
state: string;
|
|
58
|
+
health: { cell: string; monitored: boolean; firingCount: number; suppressed: boolean };
|
|
59
|
+
systems: ConsoleSystem[];
|
|
60
|
+
/**
|
|
61
|
+
* Epoch MILLISECONDS of the newest COMPLETED backup, or null if never.
|
|
62
|
+
*
|
|
63
|
+
* Milliseconds because every other instant that crosses this boundary is in
|
|
64
|
+
* milliseconds, and a single field in seconds is a factor of a thousand that
|
|
65
|
+
* no type catches. The console renders ages, so the failure mode is a
|
|
66
|
+
* plausible wrong number rather than a crash: the alerts route computed one
|
|
67
|
+
* age in the wrong unit and rendered every alert as `0m`, a fleet where
|
|
68
|
+
* nothing had been wrong for over a minute.
|
|
69
|
+
*/
|
|
70
|
+
lastBackupAt: number | null;
|
|
71
|
+
lastBackupFailed: boolean;
|
|
72
|
+
/**
|
|
73
|
+
* Nothing is checking this module.
|
|
74
|
+
*
|
|
75
|
+
* No enabled monitor, or one whose interval is manual, or one that has never
|
|
76
|
+
* run. Distinct from healthy: an unwatched module's silence is not evidence
|
|
77
|
+
* that it is fine, and `ok` and `unwatched` look identical on a dashboard
|
|
78
|
+
* that only tracks health.
|
|
79
|
+
*/
|
|
80
|
+
unwatched: boolean;
|
|
81
|
+
/**
|
|
82
|
+
* It IS checked, on a schedule, and its monitor has no escalation policy.
|
|
83
|
+
*
|
|
84
|
+
* The alert is raised and reaches nobody. Worse than unwatched in one
|
|
85
|
+
* respect, because it looks monitored, so it is reported separately rather
|
|
86
|
+
* than folded into the health cell.
|
|
87
|
+
*/
|
|
88
|
+
pagesNobody: boolean;
|
|
89
|
+
/**
|
|
90
|
+
* The newest successful backup is older than this module's cadence allows.
|
|
91
|
+
*
|
|
92
|
+
* Resolved HERE, through the same accessor the backup sweep and the drift
|
|
93
|
+
* audit use, so the three cannot disagree about what a module's cadence is.
|
|
94
|
+
* False for a module with no `on_backup` hook: there is nothing to run, so it
|
|
95
|
+
* is not overdue.
|
|
96
|
+
*/
|
|
97
|
+
backupStale: boolean;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
export interface ConsoleStatus {
|
|
101
|
+
/**
|
|
102
|
+
* The canonical zone order, most exposed first. Served rather than compiled
|
|
103
|
+
* into the console so a zone added to `NETWORK_ZONES` gets a band without a
|
|
104
|
+
* console release.
|
|
105
|
+
*/
|
|
106
|
+
zones: string[];
|
|
107
|
+
modules: ConsoleModule[];
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* The dashboard's single poll: the zone order, and every module with the two
|
|
112
|
+
* derived columns the roster shows.
|
|
113
|
+
*
|
|
114
|
+
* One call rather than three because a poll that changes nothing should cost
|
|
115
|
+
* one round trip, and because the topology, the roster and the backup column
|
|
116
|
+
* all read the same rows.
|
|
117
|
+
*/
|
|
118
|
+
export function consoleStatus(db: DbClient): ConsoleStatus {
|
|
119
|
+
const health = loadObservedHealthDetail(db);
|
|
120
|
+
const systemsByModule = new Map<string, ConsoleSystem[]>();
|
|
121
|
+
|
|
122
|
+
for (const row of db.select().from(moduleSystems).all()) {
|
|
123
|
+
const list = systemsByModule.get(row.moduleId) ?? [];
|
|
124
|
+
list.push({ hostname: row.hostname, address: row.ipv4Address, zone: row.zone });
|
|
125
|
+
systemsByModule.set(row.moduleId, list);
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
const backupByModule = latestBackupByModule(db);
|
|
129
|
+
const watch = watchFactsByModule(db);
|
|
130
|
+
const overdue = staleBackupModules(db);
|
|
131
|
+
|
|
132
|
+
const rows = db
|
|
133
|
+
.select({ id: modules.id, version: modules.version, state: modules.state })
|
|
134
|
+
.from(modules)
|
|
135
|
+
.all();
|
|
136
|
+
|
|
137
|
+
return {
|
|
138
|
+
zones: [...NETWORK_ZONES],
|
|
139
|
+
modules: rows
|
|
140
|
+
.map((row): ConsoleModule => {
|
|
141
|
+
const backup = backupByModule.get(row.id);
|
|
142
|
+
return {
|
|
143
|
+
id: row.id,
|
|
144
|
+
version: row.version,
|
|
145
|
+
state: row.state,
|
|
146
|
+
health: health.get(row.id) ?? {
|
|
147
|
+
// A module that is not deployed has nothing to observe. Saying
|
|
148
|
+
// "not observed" here would report undeployed modules as a finding.
|
|
149
|
+
cell: 'not deployed',
|
|
150
|
+
monitored: false,
|
|
151
|
+
firingCount: 0,
|
|
152
|
+
suppressed: false,
|
|
153
|
+
},
|
|
154
|
+
systems: systemsByModule.get(row.id) ?? [],
|
|
155
|
+
lastBackupAt: backup?.lastSuccessAt ?? null,
|
|
156
|
+
lastBackupFailed: backup?.lastFailed ?? false,
|
|
157
|
+
// Absent from the map means no enabled monitor at all, which is the
|
|
158
|
+
// loudest form of unwatched rather than the quietest.
|
|
159
|
+
unwatched: watch.get(row.id)?.unwatched ?? true,
|
|
160
|
+
pagesNobody: watch.get(row.id)?.pagesNobody ?? false,
|
|
161
|
+
backupStale: overdue.has(row.id),
|
|
162
|
+
};
|
|
163
|
+
})
|
|
164
|
+
.sort((a, b) => a.id.localeCompare(b.id)),
|
|
165
|
+
};
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* What celilo is doing to watch each module, if anything.
|
|
170
|
+
*
|
|
171
|
+
* Two facts, kept apart because an operator acts differently on each. A module
|
|
172
|
+
* nothing checks is invisible. A module that IS checked on a schedule and whose
|
|
173
|
+
* monitor routes to no escalation policy is worse in one respect: it looks
|
|
174
|
+
* monitored, the alert gets raised on time, and it reaches nobody.
|
|
175
|
+
*
|
|
176
|
+
* `interval manual` and `never run` both count as unwatched. A monitor that
|
|
177
|
+
* exists and has not executed produces exactly as much evidence as one that
|
|
178
|
+
* does not exist, and the console must not report the first as coverage.
|
|
179
|
+
*/
|
|
180
|
+
interface WatchFacts {
|
|
181
|
+
unwatched: boolean;
|
|
182
|
+
pagesNobody: boolean;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
function watchFactsByModule(db: DbClient): Map<string, WatchFacts> {
|
|
186
|
+
const result = new Map<string, WatchFacts>();
|
|
187
|
+
for (const monitor of db.select().from(monitors).where(eq(monitors.enabled, true)).all()) {
|
|
188
|
+
const running = monitor.intervalMinutes > 0 && monitor.lastRunAt !== null;
|
|
189
|
+
const existing = result.get(monitor.target);
|
|
190
|
+
// A module can carry more than one monitor. It is watched if ANY of them
|
|
191
|
+
// runs, and pages nobody only if every running one lacks a policy.
|
|
192
|
+
result.set(monitor.target, {
|
|
193
|
+
unwatched: (existing?.unwatched ?? true) && !running,
|
|
194
|
+
pagesNobody:
|
|
195
|
+
(existing?.pagesNobody ?? false) || (running && monitor.escalationPolicyId === null),
|
|
196
|
+
});
|
|
197
|
+
}
|
|
198
|
+
return result;
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* Modules whose newest successful backup is older than their cadence allows.
|
|
203
|
+
*
|
|
204
|
+
* Resolved through `effectiveBackupSchedule` and `backupStaleThresholdMs`, the
|
|
205
|
+
* same two functions the drift audit uses. The cadence in force comes from the
|
|
206
|
+
* operator's override and then the manifest and defaults to daily when neither
|
|
207
|
+
* says anything, so a console that worked it out again would be a second
|
|
208
|
+
* definition of "overdue" and would disagree the first time an override was
|
|
209
|
+
* set.
|
|
210
|
+
*
|
|
211
|
+
* A module with no `on_backup` hook is never stale. There is nothing to run.
|
|
212
|
+
*/
|
|
213
|
+
function staleBackupModules(db: DbClient): Set<string> {
|
|
214
|
+
const stale = new Set<string>();
|
|
215
|
+
const now = Date.now();
|
|
216
|
+
for (const info of loadBackupAuditInfo(db)) {
|
|
217
|
+
if (!moduleHasBackupHook(info.manifest)) continue;
|
|
218
|
+
const cadence = effectiveBackupSchedule(info.manifest, info.scheduleOverride);
|
|
219
|
+
const threshold = backupStaleThresholdMs(cadence);
|
|
220
|
+
// `manual` has no threshold. Opting out is a decision, not a fault.
|
|
221
|
+
if (threshold === null) continue;
|
|
222
|
+
const last = info.lastSuccessfulBackupAt;
|
|
223
|
+
if (last === null || now - last > threshold) stale.add(info.id);
|
|
224
|
+
}
|
|
225
|
+
return stale;
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
interface BackupFacts {
|
|
229
|
+
lastSuccessAt: number | null;
|
|
230
|
+
lastFailed: boolean;
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/**
|
|
234
|
+
* Newest completed backup per module, and whether the most recent ATTEMPT
|
|
235
|
+
* failed.
|
|
236
|
+
*
|
|
237
|
+
* Both, because they answer different questions and the console shows both. A
|
|
238
|
+
* module can hold a fresh successful backup and still have failed last night,
|
|
239
|
+
* and a roster that showed only the success date would call that healthy.
|
|
240
|
+
*/
|
|
241
|
+
function latestBackupByModule(db: DbClient): Map<string, BackupFacts> {
|
|
242
|
+
const result = new Map<string, BackupFacts>();
|
|
243
|
+
const rows = db
|
|
244
|
+
.select({
|
|
245
|
+
moduleId: backups.moduleId,
|
|
246
|
+
status: backups.status,
|
|
247
|
+
startedAt: backups.startedAt,
|
|
248
|
+
completedAt: backups.completedAt,
|
|
249
|
+
})
|
|
250
|
+
.from(backups)
|
|
251
|
+
.orderBy(desc(backups.startedAt))
|
|
252
|
+
.all();
|
|
253
|
+
|
|
254
|
+
for (const row of rows) {
|
|
255
|
+
if (!row.moduleId) continue; // a system-state backup belongs to no module
|
|
256
|
+
const existing = result.get(row.moduleId);
|
|
257
|
+
if (!existing) {
|
|
258
|
+
result.set(row.moduleId, {
|
|
259
|
+
lastSuccessAt:
|
|
260
|
+
row.status === 'completed' && row.completedAt ? epochMillis(row.completedAt) : null,
|
|
261
|
+
lastFailed: row.status === 'failed',
|
|
262
|
+
});
|
|
263
|
+
continue;
|
|
264
|
+
}
|
|
265
|
+
if (existing.lastSuccessAt === null && row.status === 'completed' && row.completedAt) {
|
|
266
|
+
existing.lastSuccessAt = epochMillis(row.completedAt);
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
return result;
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* Milliseconds, deliberately, and named so nobody has to guess.
|
|
274
|
+
*
|
|
275
|
+
* This was `epochSeconds`, and the protocol on the other side of the boundary
|
|
276
|
+
* documents the same field as milliseconds. The two never met because the
|
|
277
|
+
* console server does not exist yet, so the factor of a thousand sat there
|
|
278
|
+
* looking like working code. It is the exact shape of the bug that rendered
|
|
279
|
+
* every alert as `0m`.
|
|
280
|
+
*/
|
|
281
|
+
function epochMillis(value: Date): number {
|
|
282
|
+
return value.getTime();
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/**
|
|
286
|
+
* Which providers a module has actually called into, by capability.
|
|
287
|
+
*
|
|
288
|
+
* The difference between a manifest and a fleet. A manifest says what a module
|
|
289
|
+
* CAN consume; this says what it resolved to and reached for. `tango-nexus`
|
|
290
|
+
* declares four optional capabilities and is bound to one, and until
|
|
291
|
+
* celilo#1072 landed there was no way to tell those apart.
|
|
292
|
+
*
|
|
293
|
+
* The console previously inferred this: a REQUIRED capability of a deployed
|
|
294
|
+
* module must have bound, because the deploy would have failed otherwise. That
|
|
295
|
+
* was the only sound inference available and it is now an approximation, so it
|
|
296
|
+
* is gone rather than kept as a fallback (Rule 3.9). A module with no rows here
|
|
297
|
+
* has called into nothing, which is a real answer and not a missing one.
|
|
298
|
+
*/
|
|
299
|
+
export function loadBindings(db: DbClient, moduleId: string): Map<string, string> {
|
|
300
|
+
return new Map(
|
|
301
|
+
listCapabilityBindings(db, moduleId).map((b) => [b.capabilityName, b.providerModuleId]),
|
|
302
|
+
);
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
/** Everything the closure walk needs, read in one place. */
|
|
306
|
+
export function loadClosureInputs(db: DbClient): {
|
|
307
|
+
manifests: Map<string, ConsumedCapabilities>;
|
|
308
|
+
providerStates: { moduleId: string; state: string }[];
|
|
309
|
+
} {
|
|
310
|
+
return {
|
|
311
|
+
manifests: new Map(
|
|
312
|
+
db
|
|
313
|
+
.select({ id: modules.id, manifestData: modules.manifestData })
|
|
314
|
+
.from(modules)
|
|
315
|
+
.all()
|
|
316
|
+
.map((row) => [row.id, CONSUMED_CAPABILITIES.parse(row.manifestData)] as const),
|
|
317
|
+
),
|
|
318
|
+
providerStates: db.select({ moduleId: modules.id, state: modules.state }).from(modules).all(),
|
|
319
|
+
};
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
/** A module's row, or undefined. Used to reject a closure request for a module that is gone. */
|
|
323
|
+
export function moduleExists(db: DbClient, moduleId: string): boolean {
|
|
324
|
+
return (
|
|
325
|
+
db.select({ id: modules.id }).from(modules).where(eq(modules.id, moduleId)).all().length > 0
|
|
326
|
+
);
|
|
327
|
+
}
|
package/src/db/schema.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { INSTANCE_STATES, type InstanceState } from '@celilo/capabilities';
|
|
1
2
|
import { sql } from 'drizzle-orm';
|
|
2
3
|
import {
|
|
3
4
|
index,
|
|
@@ -300,22 +301,14 @@ export const moduleIntegrity = sqliteTable('module_integrity', {
|
|
|
300
301
|
});
|
|
301
302
|
|
|
302
303
|
/**
|
|
303
|
-
* Lifecycle states an instance moves through
|
|
304
|
-
* `list` so a parent can compare desired against observed (design D6).
|
|
304
|
+
* Lifecycle states an instance moves through.
|
|
305
305
|
*
|
|
306
|
-
*
|
|
307
|
-
*
|
|
308
|
-
*
|
|
306
|
+
* Re-exported from `@celilo/capabilities` rather than redefined here. `list`
|
|
307
|
+
* PROMISES these to a caller's reconcile loop, so the vocabulary belongs to the
|
|
308
|
+
* contract; a second copy in the schema is how a column and an interface come
|
|
309
|
+
* to disagree about what `failed` means.
|
|
309
310
|
*/
|
|
310
|
-
export
|
|
311
|
-
'pending',
|
|
312
|
-
'provisioning',
|
|
313
|
-
'ready',
|
|
314
|
-
'failed',
|
|
315
|
-
'destroying',
|
|
316
|
-
] as const;
|
|
317
|
-
|
|
318
|
-
export type InstanceState = (typeof INSTANCE_STATES)[number];
|
|
311
|
+
export { INSTANCE_STATES, type InstanceState };
|
|
319
312
|
|
|
320
313
|
/**
|
|
321
314
|
* Instances of a submodule (openspec/changes/submodules, D2 and D4).
|
|
@@ -492,6 +485,18 @@ export const moduleBuilds = sqliteTable('module_builds', {
|
|
|
492
485
|
* that is someone else's cloud, this is a network the fleet's own firewall
|
|
493
486
|
* holds a leg on and must translate for.
|
|
494
487
|
*/
|
|
488
|
+
/**
|
|
489
|
+
* ⚠️ ADDING A ZONE ALSO MEANS TOUCHING THE WEB CONSOLE.
|
|
490
|
+
*
|
|
491
|
+
* A zone comes into being here and nowhere else. How it is DRAWN — its accent
|
|
492
|
+
* colour and the one-line description a person reads under its name — lives in
|
|
493
|
+
* `apps/console/src/zones.ts`, because that is presentation and this is not.
|
|
494
|
+
*
|
|
495
|
+
* A zone added here and missed there draws grey with "no description yet",
|
|
496
|
+
* which is legible and wrong. `apps/console/tests/zones.test.ts` fails until
|
|
497
|
+
* the entry exists, so the pull request that adds the zone is the one that
|
|
498
|
+
* finds out, rather than the release that ships it.
|
|
499
|
+
*/
|
|
495
500
|
export const NETWORK_ZONES = [
|
|
496
501
|
'isp-transit',
|
|
497
502
|
'internal',
|
package/src/hooks/broker.test.ts
CHANGED
|
@@ -58,12 +58,10 @@ async function runCapabilityHook(): Promise<Record<string, unknown>> {
|
|
|
58
58
|
stateDir: dir,
|
|
59
59
|
capabilities: demoCapabilities(),
|
|
60
60
|
};
|
|
61
|
-
return await executeHookScript(
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
30_000,
|
|
66
|
-
);
|
|
61
|
+
return await executeHookScript(join(FIXTURES, 'capability-calling-hook.ts'), context, {
|
|
62
|
+
timeoutMs: 30_000,
|
|
63
|
+
idleTimeoutMs: 30_000,
|
|
64
|
+
});
|
|
67
65
|
} finally {
|
|
68
66
|
rmSync(dir, { recursive: true, force: true });
|
|
69
67
|
}
|