@celilo/cli 0.14.4 → 0.16.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/CELILO_CORE_MODULES.md +1 -1
  2. package/CELILO_SUBSYSTEMS.md +19 -2
  3. package/drizzle/0018_drop_alert_policy_snapshot.sql +46 -0
  4. package/drizzle/meta/_journal.json +8 -1
  5. package/package.json +3 -3
  6. package/src/cli/commands/alerts-list.ts +10 -0
  7. package/src/cli/commands/alerts-poll.ts +12 -6
  8. package/src/cli/commands/alerts-sweep.ts +22 -81
  9. package/src/cli/commands/backup-sweep.ts +65 -0
  10. package/src/cli/commands/module-config.test.ts +77 -1
  11. package/src/cli/commands/module-config.ts +45 -3
  12. package/src/cli/commands/module-journal.test.ts +47 -0
  13. package/src/cli/commands/module-journal.ts +98 -0
  14. package/src/cli/commands/module-operations.test.ts +93 -0
  15. package/src/cli/commands/module-operations.ts +134 -0
  16. package/src/cli/commands/module-upgrade.test.ts +32 -20
  17. package/src/cli/commands/module-upgrade.ts +37 -32
  18. package/src/cli/commands/monitor.ts +26 -6
  19. package/src/cli/commands/system-audit.ts +3 -30
  20. package/src/cli/completion.ts +20 -1
  21. package/src/cli/generate-zsh-completion.ts +4 -0
  22. package/src/cli/index.ts +14 -0
  23. package/src/db/schema.ts +5 -3
  24. package/src/manifest/schema.ts +4 -1
  25. package/src/module/packaging/build.ts +4 -0
  26. package/src/services/alerting/builtin-source.ts +17 -2
  27. package/src/services/alerting/delivery-loop.test.ts +5 -1
  28. package/src/services/alerting/format.test.ts +0 -1
  29. package/src/services/alerting/inbound-poller.test.ts +235 -8
  30. package/src/services/alerting/inbound-poller.ts +95 -34
  31. package/src/services/alerting/inbound.test.ts +213 -2
  32. package/src/services/alerting/inbound.ts +161 -32
  33. package/src/services/alerting/interview-responder.test.ts +0 -32
  34. package/src/services/alerting/interview-responder.ts +6 -17
  35. package/src/services/alerting/notify-deps.ts +113 -0
  36. package/src/services/alerting/run-monitor.ts +0 -1
  37. package/src/services/alerting/store.test.ts +1 -1
  38. package/src/services/alerting/store.ts +0 -2
  39. package/src/services/alerting/sweep-runner.test.ts +11 -2
  40. package/src/services/alerting/sweep-runner.ts +14 -7
  41. package/src/services/alerting/tokens.ts +39 -1
  42. package/src/services/audit/backup-source.ts +54 -0
  43. package/src/services/audit/backups.test.ts +7 -2
  44. package/src/services/audit/backups.ts +10 -18
  45. package/src/services/backup-cipher.test.ts +188 -0
  46. package/src/services/backup-cipher.ts +178 -0
  47. package/src/services/backup-create.ts +20 -30
  48. package/src/services/backup-envelope-roundtrip.test.ts +6 -26
  49. package/src/services/backup-restore.ts +10 -16
  50. package/src/services/backup-schedule.ts +35 -0
  51. package/src/services/backup-sweep.test.ts +148 -0
  52. package/src/services/backup-sweep.ts +124 -0
  53. package/src/services/deploy-posture.ts +15 -2
  54. package/src/services/module-journal.test.ts +302 -0
  55. package/src/services/module-journal.ts +160 -0
  56. package/src/services/module-operations.test.ts +67 -6
  57. package/src/services/module-operations.ts +69 -19
  58. package/src/services/module-subscriptions.test.ts +33 -2
  59. package/src/services/module-subscriptions.ts +10 -1
  60. package/src/services/module-validator/typescript-build.test.ts +20 -1
  61. package/src/services/module-validator/typescript-build.ts +9 -5
  62. package/src/services/restore-from-file.ts +6 -21
  63. package/src/templates/generator.test.ts +88 -0
  64. package/src/templates/generator.ts +119 -16
@@ -12,20 +12,52 @@
12
12
  * A message from an address matching no route is discarded and logged WITHOUT
13
13
  * a reply. Replying would confirm the number is live and that this is a celilo
14
14
  * instance, which is free reconnaissance for anyone probing.
15
+ *
16
+ * ── ORDER MATTERS: RESOLVE THE TOKEN, THEN PICK THE GRAMMAR ────────────────
17
+ *
18
+ * The same table carries alert pages and deploy questions, and they want
19
+ * OPPOSITE parsing. An alert reply is a token and at most a verb; anything
20
+ * wordier is refused, because `<token> resolve` silently acknowledging an
21
+ * alert the operator meant to escalate is the worst outcome here. An interview
22
+ * answer is a token followed by arbitrary text, because the text IS the answer.
23
+ *
24
+ * Parsing before resolving cannot serve both, and #533 is what that cost: the
25
+ * alert grammar ran first, read every real answer as a sentence, and rejected
26
+ * it before the poller could ever see it was a question. The feature was dead
27
+ * for months and the operator was told `unrecognised`, which reads as their
28
+ * typing being wrong.
29
+ *
30
+ * A delivery knows its own kind, so the token is resolved FIRST and the
31
+ * grammar chosen from what it named. Which rule applies was never a property
32
+ * of the text.
15
33
  */
16
34
 
17
35
  import { eq } from 'drizzle-orm';
18
36
  import type { DbClient } from '../../db/client';
19
37
  import { type NotificationDelivery, type Route, routes } from '../../db/schema';
20
- import { findLiveDelivery, normaliseToken } from './tokens';
38
+ import { findLiveDeliveriesByPrefix, normaliseToken } from './tokens';
21
39
 
22
40
  /**
23
- * v1 grammar: a bare token acknowledges. `snooze` and `resolve` verbs are
24
- * deferred — ack is what an operator does at 3am, and every extra verb is
25
- * another thing to mistype under stress.
41
+ * v1 grammar: a token acknowledges. `snooze` and `resolve` verbs are deferred —
42
+ * ack is what an operator does at 3am, and every extra verb is another thing to
43
+ * mistype under stress.
26
44
  *
27
- * A bare `ack` with no token is accepted when the sender has exactly one
28
- * outstanding delivery, since the token adds nothing when there is no ambiguity.
45
+ * Deliberately forgiving about everything EXCEPT what the message means. An
46
+ * earlier cut accepted `<token> ack` but not `ack <token>`, so an operator who
47
+ * read "reply BPRJEH" and typed the natural "ack BPRJEH" was refused — the verb
48
+ * was parsed as the token, failed the length check, and the reply was rejected.
49
+ * That is a real page going unacknowledged over word order, so:
50
+ *
51
+ * - the ack verb may come BEFORE the token, AFTER it, or not at all
52
+ * - case never matters, for the verb or the token
53
+ * - the token may be shortened to any unambiguous prefix (see below)
54
+ *
55
+ * `token` here is a CANDIDATE, not necessarily a whole token: resolution
56
+ * against what is actually outstanding happens in `interpretInbound`, because
57
+ * only the database knows which prefixes are unique right now.
58
+ *
59
+ * A bare ack with no token at all is accepted when the sender has exactly one
60
+ * outstanding delivery, since a token adds nothing when there is no ambiguity.
29
61
  */
30
62
  export type InboundIntent =
31
63
  | { kind: 'ack'; token: string }
@@ -33,37 +65,87 @@ export type InboundIntent =
33
65
  | { kind: 'unrecognised' };
34
66
 
35
67
  const ACK_VERB = /^(ack|ok|k|👍)$/iu;
36
- const TOKEN_SHAPE = /^[0-9ABCDEFGHJKMNPQRSTVWXYZ]{6}$/;
68
+ /**
69
+ * One to six characters of the token alphabet. Not anchored to the full
70
+ * length: a prefix is legal input, and whether it identifies something is a
71
+ * question for the delivery table, not for a regex.
72
+ */
73
+ const TOKEN_SHAPE = /^[0-9ABCDEFGHJKMNPQRSTVWXYZ]{1,6}$/;
37
74
 
38
75
  export function parseInbound(body: string): InboundIntent {
39
- const trimmed = body.trim();
40
- if (!trimmed) return { kind: 'unrecognised' };
76
+ const words = body.trim().split(/\s+/).filter(Boolean);
77
+ if (words.length === 0) return { kind: 'unrecognised' };
78
+
79
+ // Strip ack synonyms wherever they appear. What remains must be the token,
80
+ // or nothing at all.
81
+ const remainder = words.filter((word) => !ACK_VERB.test(word));
82
+ if (remainder.length === 0) return { kind: 'bare_ack' };
83
+
84
+ // More than one non-verb word is not a token with politeness around it, it
85
+ // is a sentence — and a sentence may be a request celilo does not implement.
86
+ // Guessing at it risks acknowledging an alert the operator meant to escalate:
87
+ // an earlier cut let `<token> resolve` silently ACK, so the operator believed
88
+ // they had cleared an alert that was still firing. Doing nothing and saying
89
+ // so is strictly better than doing the wrong thing quietly.
90
+ if (remainder.length > 1) return { kind: 'unrecognised' };
91
+
92
+ const token = normaliseToken(remainder[0]);
93
+ if (!TOKEN_SHAPE.test(token)) return { kind: 'unrecognised' };
41
94
 
42
- if (ACK_VERB.test(trimmed)) return { kind: 'bare_ack' };
95
+ return { kind: 'ack', token };
96
+ }
43
97
 
44
- const [first, ...rest] = trimmed.split(/\s+/);
45
- const token = normaliseToken(first);
46
- if (!TOKEN_SHAPE.test(token)) return { kind: 'unrecognised' };
98
+ /** A message that leads with a token, and whatever text followed it. */
99
+ export interface TokenLedMessage {
100
+ token: string;
101
+ /** Everything after the token, trimmed. Empty when the token stood alone. */
102
+ rest: string;
103
+ }
47
104
 
48
- // A verb after the token must be an ack synonym or nothing. An earlier cut
49
- // ignored the verb entirely, which meant `<token> resolve` silently
50
- // ACKNOWLEDGED — the operator believes they cleared the alert and it is
51
- // still firing. Doing nothing and saying so is strictly better than doing
52
- // the wrong thing quietly.
53
- //
54
- // `resolve` and `silence` are terminal operations, not reply verbs (design
55
- // D10): every extra verb is another thing to mistype at 3am.
56
- if (rest.length > 0 && !ACK_VERB.test(rest.join(' '))) {
57
- return { kind: 'unrecognised' };
58
- }
105
+ /**
106
+ * Split a message that LEADS with a token, or null if it does not.
107
+ *
108
+ * This is the interview grammar and it is deliberately not the alert one: the
109
+ * page says `reply <TOKEN> <value>`, and the value is whatever follows,
110
+ * verbatim. It may contain spaces, punctuation, or a word that happens to look
111
+ * like a verb second-guessing it would corrupt exactly the values an operator
112
+ * cannot easily retype.
113
+ *
114
+ * `parseInbound` cannot serve both. Its "more than one non-verb word is a
115
+ * sentence" rule is what protects an ALERT from `<token> resolve`, and it is
116
+ * also what made every real interview answer unrecognised (#533): a value is
117
+ * a sentence by construction. Which rule applies is not a property of the text
118
+ * — it is a property of what the token names, and only the delivery table
119
+ * knows that. So both readings are produced here and `interpretInbound`
120
+ * chooses between them AFTER resolving the token.
121
+ */
122
+ export function parseTokenLed(body: string): TokenLedMessage | null {
123
+ const trimmed = body.trim();
124
+ if (!trimmed) return null;
59
125
 
60
- return { kind: 'ack', token };
126
+ const boundary = trimmed.search(/\s/);
127
+ const head = boundary === -1 ? trimmed : trimmed.slice(0, boundary);
128
+ const token = normaliseToken(head);
129
+ if (!TOKEN_SHAPE.test(token)) return null;
130
+
131
+ return { token, rest: boundary === -1 ? '' : trimmed.slice(boundary).trim() };
61
132
  }
62
133
 
63
134
  export type InboundOutcome =
64
135
  | { action: 'ack'; delivery: NotificationDelivery; route: Route }
136
+ /** An answer to a deploy question. `value` is the operator's text, verbatim. */
137
+ | { action: 'answer'; delivery: NotificationDelivery; route: Route; value: string }
65
138
  | { action: 'ignored'; reason: 'unknown_sender' }
66
- | { action: 'rejected'; reason: 'unknown_token' | 'wrong_sender' | 'ambiguous' | 'unrecognised' };
139
+ | {
140
+ action: 'rejected';
141
+ reason:
142
+ | 'unknown_token'
143
+ | 'wrong_sender'
144
+ | 'ambiguous'
145
+ | 'unrecognised'
146
+ /** A question was named, but no value was supplied to answer it with. */
147
+ | 'needs_value';
148
+ };
67
149
 
68
150
  function routeForAddress(db: DbClient, senderAddress: string): Route | undefined {
69
151
  return db.select().from(routes).where(eq(routes.address, senderAddress)).get();
@@ -91,10 +173,9 @@ export function interpretInbound(db: DbClient, context: InboundContext): Inbound
91
173
 
92
174
  const intent = parseInbound(context.body);
93
175
 
94
- if (intent.kind === 'unrecognised') {
95
- return { action: 'rejected', reason: 'unrecognised' };
96
- }
97
-
176
+ // A bare ack names no token, so there is nothing to resolve and nothing a
177
+ // question could be answered with. It is an ALERT reply by construction —
178
+ // `outstandingForRoute` supplies alert deliveries only.
98
179
  if (intent.kind === 'bare_ack') {
99
180
  const outstanding = context.outstandingForRoute(route.id);
100
181
  if (outstanding.length === 0) return { action: 'rejected', reason: 'unknown_token' };
@@ -102,11 +183,59 @@ export function interpretInbound(db: DbClient, context: InboundContext): Inbound
102
183
  return { action: 'ack', delivery: outstanding[0], route };
103
184
  }
104
185
 
105
- const delivery = findLiveDelivery(db, intent.token, context.now);
106
- if (!delivery) return { action: 'rejected', reason: 'unknown_token' };
186
+ // Both readings of the text, neither yet privileged. The alert reading is
187
+ // the more constrained one, so it names the token when it applies; otherwise
188
+ // a leading token does.
189
+ const led = parseTokenLed(context.body);
190
+ const namesAToken = intent.kind === 'ack';
191
+ const candidate = namesAToken ? intent.token : led?.token;
192
+
193
+ // Nothing token-shaped anywhere: not a reply celilo can act on.
194
+ if (!candidate) return { action: 'rejected', reason: 'unrecognised' };
195
+
196
+ // A prefix is resolved against what is LIVE right now, so how much of the
197
+ // token an operator must type depends on what is actually outstanding — one
198
+ // alert, one character. Matching is deliberately global rather than scoped to
199
+ // this route: it keeps `wrong_sender` a distinguishable answer below, and a
200
+ // globally-unique prefix is a stricter bar than a per-route one.
201
+ const matches = findLiveDeliveriesByPrefix(db, candidate, context.now);
202
+ if (matches.length === 0) {
203
+ // Ordinary words are token-shaped surprisingly often — the alphabet is
204
+ // most of the Latin one, so `what is going on` leads with a perfectly
205
+ // well-formed `WHAT`. Only call it a token the operator got WRONG when the
206
+ // alert grammar agreed it was one; otherwise this is a sentence, and
207
+ // saying `unknown_token` would send them hunting for a typo they did not
208
+ // make.
209
+ return { action: 'rejected', reason: namesAToken ? 'unknown_token' : 'unrecognised' };
210
+ }
211
+
212
+ // Ambiguous is its own answer, never a guess. Taking the first match would
213
+ // acknowledge an alert the operator did not name.
214
+ if (matches.length > 1) return { action: 'rejected', reason: 'ambiguous' };
215
+
216
+ const delivery = matches[0];
107
217
 
108
218
  // Factor two: a valid token replayed from a different number is refused.
219
+ // Checked before the grammar so a stranger learns nothing about which of the
220
+ // two a token names.
109
221
  if (delivery.routeId !== route.id) return { action: 'rejected', reason: 'wrong_sender' };
110
222
 
223
+ // NOW the kind is known, so the right grammar can be applied to the body.
224
+ if (delivery.kind === 'interview') {
225
+ // The value must follow the token the question was asked with; a verb-led
226
+ // message (`ack <TOKEN>`) names no value and is not an answer.
227
+ if (!led || led.token !== candidate) return { action: 'rejected', reason: 'unrecognised' };
228
+ // Named the question but supplied nothing to answer it with. Distinct from
229
+ // `unrecognised`: the operator got the token right and stopped too soon,
230
+ // and telling them that is the difference between retyping six characters
231
+ // and giving up.
232
+ if (!led.rest) return { action: 'rejected', reason: 'needs_value' };
233
+ return { action: 'answer', delivery, route, value: led.rest };
234
+ }
235
+
236
+ // An alert. The strict grammar applies, which is what keeps `<token> resolve`
237
+ // from silently acknowledging something the operator meant to escalate.
238
+ if (intent.kind !== 'ack') return { action: 'rejected', reason: 'unrecognised' };
239
+
111
240
  return { action: 'ack', delivery, route };
112
241
  }
@@ -4,7 +4,6 @@ import {
4
4
  decideInterviewDelivery,
5
5
  describeQuestion,
6
6
  isRefusedFamily,
7
- parseInterviewAnswer,
8
7
  } from './interview-responder';
9
8
 
10
9
  const headless = { hasTty: false, hasBidirectionalRoute: true };
@@ -98,37 +97,6 @@ describe('composeInterviewBody', () => {
98
97
  });
99
98
  });
100
99
 
101
- describe('parseInterviewAnswer', () => {
102
- test('takes everything after the token as the answer', () => {
103
- expect(parseInterviewAnswer('K7QM2X www.example.com', 'K7QM2X')).toBe('www.example.com');
104
- });
105
-
106
- test('is case-insensitive on the token', () => {
107
- expect(parseInterviewAnswer('k7qm2x www.example.com', 'K7QM2X')).toBe('www.example.com');
108
- });
109
-
110
- // An answer may legitimately contain spaces and punctuation. Second-guessing
111
- // it would corrupt exactly the values that are painful to retype.
112
- test('preserves an answer containing spaces and punctuation verbatim', () => {
113
- expect(parseInterviewAnswer('K7QM2X my value: with, punctuation', 'K7QM2X')).toBe(
114
- 'my value: with, punctuation',
115
- );
116
- });
117
-
118
- test('an answer that looks like a verb is still an answer', () => {
119
- expect(parseInterviewAnswer('K7QM2X resolve', 'K7QM2X')).toBe('resolve');
120
- });
121
-
122
- test('a token with no value is not an answer', () => {
123
- expect(parseInterviewAnswer('K7QM2X', 'K7QM2X')).toBeNull();
124
- expect(parseInterviewAnswer('K7QM2X ', 'K7QM2X')).toBeNull();
125
- });
126
-
127
- test('a different token does not match', () => {
128
- expect(parseInterviewAnswer('ZZZZZZ value', 'K7QM2X')).toBeNull();
129
- });
130
- });
131
-
132
100
  /**
133
101
  * The families disagree on payload shape, and reading the wrong field means an
134
102
  * operator gets the raw event type instead of the question.
@@ -136,23 +136,12 @@ export function composeInterviewBody(input: {
136
136
  return lines.join('\n');
137
137
  }
138
138
 
139
- /**
140
- * Extract the answer from an inbound reply.
141
- *
142
- * Everything after the token is the value, verbatim and unparsed — an answer
143
- * may legitimately contain spaces, punctuation, or something that looks like a
144
- * verb, and second-guessing it would corrupt exactly the values an operator
145
- * cannot easily retype.
146
- */
147
- export function parseInterviewAnswer(body: string, token: string): string | null {
148
- const trimmed = body.trim();
149
- const upper = trimmed.toUpperCase();
150
- const normalisedToken = token.toUpperCase();
151
- if (!upper.startsWith(normalisedToken)) return null;
152
-
153
- const answer = trimmed.slice(token.length).trim();
154
- return answer.length > 0 ? answer : null;
155
- }
139
+ // `parseInterviewAnswer` lived here and is gone. It was correct and had the
140
+ // tests to prove it, and it was never reachable: the caller only ran it for a
141
+ // body the ALERT grammar had already accepted, which no real answer ever was
142
+ // (#533). Extraction now happens in `interpretInbound` the one place that
143
+ // knows a delivery's kind before it parses so the reply grammar exists once
144
+ // rather than in two halves that could disagree.
156
145
 
157
146
  /** Severity an interview question is delivered at — never a page-worthy one. */
158
147
  export const INTERVIEW_SEVERITY: AlertSeverity = 'warning';
@@ -0,0 +1,113 @@
1
+ /**
2
+ * Resolving an alert to the people it should page.
3
+ *
4
+ * The escalation policy is a property of the MONITOR, and it is read here on
5
+ * every sweep rather than copied onto the alert when the alert is created.
6
+ * That ordering is the whole point: an operator only reaches for
7
+ * `escalation-policy assign` once something is already firing and reaching
8
+ * nobody, so a policy that only applied to alerts created afterwards would be
9
+ * a no-op for the one alert they were trying to fix (#481).
10
+ *
11
+ * The cost of resolving late is that a policy edited mid-escalation changes
12
+ * where the remaining steps go. That is the behaviour an operator asks for
13
+ * when they edit it — the alert's own progress through the chain
14
+ * (`escalationStep`, `nextEscalationAt`) is what stays on the alert.
15
+ *
16
+ * Lives here rather than in the sweep command so the sweep's tests can drive
17
+ * the real resolution instead of a fake, which is how the snapshot bug
18
+ * survived: every existing test stubbed `notifyDepsFor`.
19
+ */
20
+
21
+ import { eq } from 'drizzle-orm';
22
+ import type { DbClient } from '../../db/client';
23
+ import { type Alert, type EscalationPolicy, escalationPolicies, monitors } from '../../db/schema';
24
+ import type { NotifyDeps } from './notifier';
25
+ import { listPeople, listPolicySteps, listRoutes } from './people';
26
+ import { mintDelivery } from './tokens';
27
+ import { loadNotificationTransport } from './transport-loader';
28
+
29
+ /**
30
+ * How long a reply token stays valid. A day is long enough that someone who
31
+ * sees a page overnight can still act on it in the morning, and short enough
32
+ * that a token found later is useless.
33
+ */
34
+ const TOKEN_TTL_MS = 24 * 60 * 60_000;
35
+
36
+ /**
37
+ * The escalation policy an alert resolves to right now, or null.
38
+ *
39
+ * The single answer to "who would this page", shared by the sweep and by every
40
+ * command that has to show it. Two readers computing it separately is how a
41
+ * CLI ends up confidently reporting a policy the sweep does not use.
42
+ */
43
+ export function policyForAlert(
44
+ db: DbClient,
45
+ alert: Pick<Alert, 'monitorId'>,
46
+ ): EscalationPolicy | null {
47
+ const monitor = db.select().from(monitors).where(eq(monitors.id, alert.monitorId)).get();
48
+ if (!monitor?.escalationPolicyId) return null;
49
+ return (
50
+ db
51
+ .select()
52
+ .from(escalationPolicies)
53
+ .where(eq(escalationPolicies.id, monitor.escalationPolicyId))
54
+ .get() ?? null
55
+ );
56
+ }
57
+
58
+ /**
59
+ * Assemble everything needed to page for one alert, or null when nobody can be.
60
+ *
61
+ * Returning null rather than an empty policy is deliberate: "no escalation
62
+ * policy assigned" and "policy exists but nobody is eligible" are different
63
+ * situations, and only the second is worth an escalation decision.
64
+ */
65
+ export function buildNotifyDeps(db: DbClient, alert: Alert, now: Date): NotifyDeps | null {
66
+ const policy = policyForAlert(db, alert);
67
+ if (!policy) return null;
68
+
69
+ const steps = listPolicySteps(db, policy.id).map((step) => ({
70
+ stepIndex: step.stepIndex,
71
+ routeId: step.routeId,
72
+ delayMinutes: step.delayMinutes,
73
+ }));
74
+ if (steps.length === 0) return null;
75
+
76
+ const routeRows = listRoutes(db);
77
+ const people = listPeople(db);
78
+
79
+ return {
80
+ steps,
81
+ routes: new Map(
82
+ routeRows.map((r) => [
83
+ r.id,
84
+ { id: r.id, severityFloor: r.severityFloor, enabled: r.enabled },
85
+ ]),
86
+ ),
87
+ routeDetails: new Map(routeRows.map((r) => [r.id, r])),
88
+ quietHoursByPerson: new Map(
89
+ people
90
+ .filter((p) => p.quietHoursStart && p.quietHoursEnd)
91
+ .map((p) => [
92
+ p.id,
93
+ {
94
+ personId: p.id,
95
+ start: p.quietHoursStart,
96
+ end: p.quietHoursEnd,
97
+ timezone: p.timezone,
98
+ },
99
+ ]),
100
+ ),
101
+ bypassQuietHours: policy.bypassQuietHours,
102
+ transportFor: (route) => loadNotificationTransport(db, route.transportModuleId),
103
+ mintToken: (alertId, routeId) =>
104
+ mintDelivery(db, {
105
+ kind: 'alert',
106
+ targetId: alertId,
107
+ routeId,
108
+ now,
109
+ ttlMs: TOKEN_TTL_MS,
110
+ }).token,
111
+ now,
112
+ };
113
+ }
@@ -151,7 +151,6 @@ export async function runOneMonitor(
151
151
 
152
152
  const { createdIds, resolvedIds } = applyReconcileActions(db, actions, {
153
153
  monitorId: monitor.id,
154
- escalationPolicyId: monitor.escalationPolicyId,
155
154
  now,
156
155
  });
157
156
 
@@ -27,7 +27,7 @@ describe('alert store', () => {
27
27
  let dir: string;
28
28
  let db: DbClient;
29
29
 
30
- const context = { monitorId: MONITOR, escalationPolicyId: null, now: NOW };
30
+ const context = { monitorId: MONITOR, now: NOW };
31
31
 
32
32
  const create = (key: string): ReconcileAction => ({
33
33
  type: 'create',
@@ -28,7 +28,6 @@ export function loadLiveAlerts(db: DbClient, monitorId: string): LiveAlert[] {
28
28
 
29
29
  export interface ApplyContext {
30
30
  monitorId: string;
31
- escalationPolicyId: string | null;
32
31
  now: Date;
33
32
  }
34
33
 
@@ -66,7 +65,6 @@ export function applyReconcileActions(
66
65
  lastSeenAt: context.now,
67
66
  graceUntil: action.graceUntil,
68
67
  escalationStep: 0,
69
- escalationPolicyId: context.escalationPolicyId,
70
68
  message: action.message,
71
69
  details: action.details,
72
70
  })
@@ -238,7 +238,16 @@ describe('runSweep', () => {
238
238
 
239
239
  expect(liveAlerts()).toHaveLength(1);
240
240
  expect(report.notified).toBe(0);
241
- expect(report.noPolicy).toBe(1);
241
+ expect(report.noPolicy).toEqual([{ alertKey: PORT_CHECK, monitor: MODULE }]);
242
+ });
243
+
244
+ // A count told the operator something was unroutable but not WHICH thing,
245
+ // so the remedy printed next to it could not be aimed anywhere (#481).
246
+ test('the unroutable alert is named, along with the monitor to assign to', async () => {
247
+ const report = await runSweep(db, currentMonitors(), deps());
248
+
249
+ expect(report.noPolicy[0].alertKey).toBe(PORT_CHECK);
250
+ expect(report.noPolicy[0].monitor).toBe(MODULE);
242
251
  });
243
252
 
244
253
  test('escalation declining to notify records WHICH reason', async () => {
@@ -270,7 +279,7 @@ describe('runSweep', () => {
270
279
  );
271
280
 
272
281
  expect(report.notified).toBe(0);
273
- expect(report.noPolicy).toBe(0);
282
+ expect(report.noPolicy).toEqual([]);
274
283
  expect(report.skipped.within_grace).toBe(1);
275
284
  });
276
285
 
@@ -63,13 +63,16 @@ export interface SweepReport {
63
63
  failed: number;
64
64
  /**
65
65
  * Live alerts nobody is configured to be told about — no escalation policy on
66
- * the monitor, so `notifyDepsFor` returns null.
66
+ * the monitor, so `notifyDepsFor` returns null. Each carries the monitor that
67
+ * owns it, which is the thing an operator has to assign a policy TO.
67
68
  *
68
- * Counted rather than skipped in silence: "nothing needed sending" and "an
69
- * alert is firing and no policy points at anyone" are opposite situations that
70
- * previously rendered identically as `0 notified`.
69
+ * Identified rather than merely counted: `1 no-policy` told an operator that
70
+ * something was unreachable but not WHICH thing, so the remedy the sweep
71
+ * printed alongside it could not be aimed at anything. Assigning the policy to
72
+ * all eighteen monitors then changed nothing observable (#481). A bare count
73
+ * is only half a step better than the silence it replaced.
71
74
  */
72
- noPolicy: number;
75
+ noPolicy: { alertKey: string; monitor: string }[];
73
76
  /**
74
77
  * Deliveries escalation declined, keyed by its reason (`within_grace`,
75
78
  * `no_eligible_route`, …).
@@ -113,7 +116,7 @@ export async function runSweep(
113
116
  deferred: 0,
114
117
  deferredDelivered: 0,
115
118
  failed: 0,
116
- noPolicy: 0,
119
+ noPolicy: [],
117
120
  skipped: {},
118
121
  failures: [],
119
122
  };
@@ -206,10 +209,14 @@ export async function runSweep(
206
209
  }
207
210
 
208
211
  // 5. Notify. Re-read: the steps above changed state under us.
212
+ const monitorTargets = new Map(monitors.map((m) => [m.id, m.target]));
209
213
  for (const alert of loadAllLiveAlerts(db)) {
210
214
  const notifyDeps = deps.notifyDepsFor(alert);
211
215
  if (!notifyDeps) {
212
- report.noPolicy++;
216
+ report.noPolicy.push({
217
+ alertKey: alert.key,
218
+ monitor: monitorTargets.get(alert.monitorId) ?? '(unknown monitor)',
219
+ });
213
220
  continue;
214
221
  }
215
222
 
@@ -13,7 +13,7 @@
13
13
  */
14
14
 
15
15
  import { randomInt, randomUUID } from 'node:crypto';
16
- import { and, eq, gt, isNull } from 'drizzle-orm';
16
+ import { and, eq, gt, isNull, like } from 'drizzle-orm';
17
17
  import type { DbClient } from '../../db/client';
18
18
  import { type NotificationDelivery, notificationDeliveries } from '../../db/schema';
19
19
 
@@ -100,6 +100,44 @@ export function findLiveDelivery(
100
100
  .get();
101
101
  }
102
102
 
103
+ /**
104
+ * Every live delivery whose token STARTS WITH `prefix`.
105
+ *
106
+ * An operator typing at 3am should not have to copy six characters exactly.
107
+ * They need only enough to be unambiguous, and "unambiguous" is scoped to what
108
+ * is actually outstanding right now — consumed and expired deliveries are not
109
+ * candidates, so yesterday's tokens cannot make today's prefix ambiguous.
110
+ *
111
+ * Returning every match rather than the first is the point: the caller must be
112
+ * able to tell "no such token" from "you were ambiguous, be more specific".
113
+ * Silently taking the first match would acknowledge an alert the operator did
114
+ * not mean, which is worse than asking again.
115
+ *
116
+ * The shortening does not weaken authentication in the way it first appears.
117
+ * The token was never a secret on its own — the sender address is the other
118
+ * factor and is unaffected — and the search space here is not 32^6 but the
119
+ * handful of alerts live at this moment.
120
+ */
121
+ export function findLiveDeliveriesByPrefix(
122
+ db: DbClient,
123
+ prefix: string,
124
+ now: Date,
125
+ ): NotificationDelivery[] {
126
+ const normalised = normaliseToken(prefix);
127
+ if (!normalised) return [];
128
+ return db
129
+ .select()
130
+ .from(notificationDeliveries)
131
+ .where(
132
+ and(
133
+ like(notificationDeliveries.token, `${normalised}%`),
134
+ isNull(notificationDeliveries.consumedAt),
135
+ gt(notificationDeliveries.expiresAt, now),
136
+ ),
137
+ )
138
+ .all();
139
+ }
140
+
103
141
  export function consumeDelivery(db: DbClient, deliveryId: string, now: Date): void {
104
142
  db.update(notificationDeliveries)
105
143
  .set({ consumedAt: now })
@@ -0,0 +1,54 @@
1
+ /**
2
+ * Reading the roster the backup-freshness check runs against.
3
+ *
4
+ * Split out from the check itself so the decision — is this backup too
5
+ * old — stays pure, while the queries that feed it live here (Rule 2.3).
6
+ * Shared by `celilo system audit` and by the scheduled `backups` monitor,
7
+ * so both judge the same fleet from the same data.
8
+ */
9
+
10
+ import { eq } from 'drizzle-orm';
11
+ import type { DbClient } from '../../db/client';
12
+ import { backups, modules } from '../../db/schema';
13
+ import type { ModuleManifest } from '../../manifest/schema';
14
+ import type { InstalledModuleBackupInfo } from './backups';
15
+
16
+ const DEPLOYED_STATES = ['INSTALLED', 'VERIFIED'];
17
+
18
+ /**
19
+ * Most recent COMPLETED backup per module, in epoch ms.
20
+ *
21
+ * The `backups` table is added via inline ALTER statements in
22
+ * db/client.ts for upgraded databases, so a freshly-initialized DB built
23
+ * from the drizzle journal alone may not have it yet. Absence means "no
24
+ * backups recorded", never a crashed audit.
25
+ */
26
+ function latestSuccessfulBackupByModule(db: DbClient): Map<string, number> {
27
+ const latest = new Map<string, number>();
28
+ try {
29
+ for (const backup of db.select().from(backups).where(eq(backups.status, 'completed')).all()) {
30
+ if (!backup.moduleId || !backup.completedAt) continue;
31
+ const at = backup.completedAt.getTime();
32
+ const previous = latest.get(backup.moduleId);
33
+ if (previous === undefined || at > previous) latest.set(backup.moduleId, at);
34
+ }
35
+ } catch {
36
+ // Table missing — leave the map empty.
37
+ }
38
+ return latest;
39
+ }
40
+
41
+ export function loadBackupAuditInfo(db: DbClient): InstalledModuleBackupInfo[] {
42
+ const latest = latestSuccessfulBackupByModule(db);
43
+ return db
44
+ .select()
45
+ .from(modules)
46
+ .all()
47
+ .filter((module) => DEPLOYED_STATES.includes(module.state))
48
+ .map((module) => ({
49
+ id: module.id,
50
+ state: module.state,
51
+ manifest: module.manifestData as ModuleManifest,
52
+ lastSuccessfulBackupAt: latest.get(module.id) ?? null,
53
+ }));
54
+ }