@cosmicdrift/kumiko-bundled-features 0.210.0 → 0.212.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/package.json +8 -8
  2. package/src/admin-shell/feature.ts +2 -0
  3. package/src/admin-shell/web/platform-overview-screen.tsx +1 -0
  4. package/src/admin-shell/web/tenant-overview-screen.tsx +1 -0
  5. package/src/audit/__tests__/audit-log-screen.test.tsx +228 -0
  6. package/src/audit/constants.ts +3 -0
  7. package/src/audit/feature.ts +2 -0
  8. package/src/audit/i18n.ts +1 -0
  9. package/src/audit/web/audit-log-detail-screen.tsx +28 -6
  10. package/src/audit/web/audit-log-screen.tsx +51 -23
  11. package/src/auth-email-password/feature.ts +1 -0
  12. package/src/auth-email-password/web/auth-form-primitives.tsx +2 -0
  13. package/src/auth-email-password/web/auth-gate.tsx +1 -0
  14. package/src/auth-email-password/web/confirm-account-unlock-screen.tsx +1 -0
  15. package/src/auth-email-password/web/invite-accept-screen.tsx +3 -0
  16. package/src/auth-email-password/web/session.tsx +1 -0
  17. package/src/auth-email-password/web/verify-email-screen.tsx +1 -0
  18. package/src/auth-mfa/feature.ts +1 -0
  19. package/src/auth-mfa/web/mfa-disable-dialog.tsx +1 -0
  20. package/src/auth-mfa/web/mfa-regenerate-recovery-dialog.tsx +1 -0
  21. package/src/compliance-profiles/feature.ts +1 -5
  22. package/src/crypto-shredding/changes.json +2 -1
  23. package/src/custom-fields/__tests__/feature.test.ts +42 -0
  24. package/src/custom-fields/__tests__/field-definition-role-config.integration.test.ts +139 -0
  25. package/src/custom-fields/constants.ts +6 -0
  26. package/src/custom-fields/feature.ts +56 -10
  27. package/src/custom-fields/handlers/define-tenant-field.write.ts +7 -1
  28. package/src/custom-fields/handlers/delete-tenant-field.write.ts +54 -33
  29. package/src/custom-fields/handlers/update-tenant-field.write.ts +58 -35
  30. package/src/delivery/feature.ts +1 -3
  31. package/src/feature-toggles/feature.ts +1 -0
  32. package/src/feature-toggles/web/toggle-admin-screen.tsx +1 -0
  33. package/src/file-derivatives/changes.json +9 -1
  34. package/src/folders/web/folder-manager.tsx +1 -0
  35. package/src/jobs/__tests__/jobs-pii-kms.integration.test.ts +174 -0
  36. package/src/jobs/__tests__/jobs-stale-run-sweep.integration.test.ts +148 -0
  37. package/src/jobs/db/queries/stale-run-sweep.ts +51 -0
  38. package/src/jobs/feature.ts +25 -1
  39. package/src/jobs/handlers/detail.query.ts +13 -1
  40. package/src/jobs/handlers/stale-run-sweep.job.ts +28 -0
  41. package/src/jobs/job-run-logger.ts +68 -18
  42. package/src/jobs/web/job-run-detail-screen.tsx +1 -0
  43. package/src/page-render/web.ts +1 -0
  44. package/src/personal-access-tokens/changes.json +7 -0
  45. package/src/personal-access-tokens/feature.ts +1 -0
  46. package/src/secrets/__tests__/access-options.integration.test.ts +114 -0
  47. package/src/secrets/constants.ts +12 -0
  48. package/src/secrets/feature.ts +35 -9
  49. package/src/secrets/handlers/delete.write.ts +23 -16
  50. package/src/secrets/handlers/list.query.ts +37 -30
  51. package/src/secrets/handlers/set.write.ts +43 -36
  52. package/src/secrets/index.ts +5 -0
  53. package/src/seo/feature.ts +1 -0
  54. package/src/tags/feature.ts +1 -0
  55. package/src/tags/web/tag-chip.tsx +2 -0
  56. package/src/tags/web/tag-manager.tsx +2 -1
  57. package/src/tags/web/tag-picker.tsx +1 -0
  58. package/src/template-resolver/changes.json +9 -1
  59. package/src/template-resolver/web/client-plugin.tsx +2 -0
  60. package/src/tenant/__tests__/members-screens.boot.test.ts +79 -4
  61. package/src/tenant/__tests__/tenant-security.integration.test.ts +260 -3
  62. package/src/tenant/changes.json +6 -0
  63. package/src/tenant/constants.ts +17 -4
  64. package/src/tenant/feature.ts +28 -11
  65. package/src/tenant/handlers/team-list.query.ts +250 -0
  66. package/src/tenant/i18n.ts +20 -0
  67. package/src/tenant/screens.ts +111 -3
  68. package/src/tenant/web/client-plugin.tsx +11 -6
  69. package/src/tenant/web/index.ts +2 -2
  70. package/src/tenant/web/member-roles-cell.tsx +12 -0
  71. package/src/tenant/web/member-status-cell.tsx +23 -0
  72. package/src/tier-engine/feature.ts +2 -0
  73. package/src/tier-engine/web/tier-admin-screen.tsx +5 -5
  74. package/src/user-data-rights/feature.ts +2 -0
  75. package/src/user-data-rights/web/privacy-center-screen.tsx +2 -1
  76. package/src/user-profile/i18n.ts +1 -0
  77. package/src/user-profile/web/profile-screen.tsx +1 -1
  78. package/src/tenant/web/i18n.ts +0 -24
  79. package/src/tenant/web/members-screen.tsx +0 -270
@@ -1 +1,9 @@
1
- []
1
+ [
2
+ {
3
+ "version": "0.194.0",
4
+ "type": "breaking",
5
+ "title": "PRESET_VARIANT_NAMES removed from public exports; publicVariantQuery accepts any name.",
6
+ "detail": "The public GET /media/:fileRefId/:variant route now resolves the variant spec from the FileRef's field declaration (createImageField({ variants: {...} })) instead of a fixed spec table, so an app can declare and publicly serve its own variant names, and a field overriding a built-in preset name (thumb/card/hero/full) is served with its own spec instead of the frozen preset size. publicVariantQuery's schema now accepts any variant name (z.string().min(1).max(64)) instead of z.enum(PRESET_VARIANT_NAMES); resolution requires an exact match against the field's declared variants keys, and an unresolvable name answers 404, same as an unknown FileRef. The route's pre-DB path-param gate is now purely syntactic ([a-zA-Z0-9_-]{1,64}) instead of a name allow-list; a known-valid preset name plus a random fileRefId reached the same DB read as any other name, so the list only blocked the cheaper of two equally-costly attacks, and the actual defense is the existing per-IP rate limit and the UUID guard, both unchanged.",
7
+ "migration": "PRESET_VARIANT_NAMES only ever named the allow-list this release deletes, so there is nothing to migrate to. thumb/card/hero/full remain as ready-made specs to spread into a field's own variants."
8
+ }
9
+ ]
@@ -108,6 +108,7 @@ export function FolderManager({
108
108
  const [pendingDelete, setPendingDelete] = useState<FolderNode | null>(null);
109
109
 
110
110
  // Persist expand/collapse so a reload (F5) doesn't re-expand everything.
111
+ // kumiko-lint-ignore no-raw-hooks DOM integration: writes to window.localStorage, no framework query/mutation equivalent
111
112
  useEffect(() => {
112
113
  if (typeof window === "undefined") return;
113
114
  try {
@@ -19,11 +19,15 @@ import { createEventsTable, eventsTable } from "@cosmicdrift/kumiko-framework/ev
19
19
  import {
20
20
  createTestDb,
21
21
  createTestRedis,
22
+ setupTestStack,
22
23
  type TestDb,
23
24
  type TestRedis,
25
+ type TestStack,
26
+ TestUsers,
24
27
  unsafePushTables,
25
28
  } from "@cosmicdrift/kumiko-framework/stack";
26
29
  import { resetPiiSubjectKmsForTests, resetTestTables } from "@cosmicdrift/kumiko-framework/testing";
30
+ import { JobQueries } from "../constants";
27
31
  import { createJobsFeature } from "../feature";
28
32
  import { createJobRunLogger } from "../job-run-logger";
29
33
  import { jobRunLogsTable, jobRunsTable } from "../job-run-table";
@@ -120,3 +124,173 @@ describe("jobs run-started payload under KMS", () => {
120
124
  expect(row?.["payload"]).toBe(SECRET_PAYLOAD);
121
125
  });
122
126
  });
127
+
128
+ // #2247: store_job_run_logs is the same unmanaged direct-write path as
129
+ // jobRunsTable.payload above — onJobComplete/onJobFailed batch-insert log
130
+ // lines straight from the BullMQ callback, with no event-PII catalog to
131
+ // lean on. Same subject (triggeredById), same encrypt-under-DEK treatment,
132
+ // same skip rules (null subject / no KMS stay plaintext).
133
+ describe("jobs run-completed/-failed log messages under KMS (#2247)", () => {
134
+ const SECRET_LOG = "user export contained iban DE89370400440532013000";
135
+
136
+ async function runIdFor(bullJobId: string): Promise<string> {
137
+ const row = await fetchOne(testDb.db, jobRunsTable, { bullJobId });
138
+ return String(row?.["id"]);
139
+ }
140
+
141
+ test("onJobComplete: stored log message is ciphertext, decrypts to original", async () => {
142
+ await logger.onJobStart?.("app:job:export", "bull-c1", { triggeredById: USER_ID });
143
+ await logger.onJobComplete?.("app:job:export", "bull-c1", 42, [
144
+ { level: "info", message: SECRET_LOG, timestamp: Temporal.Now.instant() },
145
+ ]);
146
+
147
+ const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-c1") });
148
+ expect(logs).toHaveLength(1);
149
+ expect(isPiiCiphertext(logs[0]?.["message"])).toBe(true);
150
+ expect(String(logs[0]?.["message"])).toContain(`user:${USER_ID}`);
151
+
152
+ const back = await decryptPiiFieldValues({ message: logs[0]?.["message"] }, ["message"], kms, {
153
+ requestId: "t",
154
+ });
155
+ expect(back["message"]).toBe(SECRET_LOG);
156
+ });
157
+
158
+ test("onJobFailed: stored log message is ciphertext, decrypts to original", async () => {
159
+ await logger.onJobStart?.("app:job:export", "bull-f1", { triggeredById: USER_ID });
160
+ await logger.onJobFailed?.("app:job:export", "bull-f1", "boom", [
161
+ { level: "error", message: SECRET_LOG, timestamp: Temporal.Now.instant() },
162
+ ]);
163
+
164
+ const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-f1") });
165
+ expect(logs).toHaveLength(1);
166
+ expect(isPiiCiphertext(logs[0]?.["message"])).toBe(true);
167
+
168
+ const back = await decryptPiiFieldValues({ message: logs[0]?.["message"] }, ["message"], kms, {
169
+ requestId: "t",
170
+ });
171
+ expect(back["message"]).toBe(SECRET_LOG);
172
+ });
173
+
174
+ test("erase subject key after write → completed log message decrypts to [[erased]]", async () => {
175
+ await logger.onJobStart?.("app:job:export", "bull-c2", { triggeredById: USER_ID });
176
+ await logger.onJobComplete?.("app:job:export", "bull-c2", 10, [
177
+ { level: "info", message: SECRET_LOG, timestamp: Temporal.Now.instant() },
178
+ ]);
179
+ const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-c2") });
180
+
181
+ await kms.eraseKey({ kind: "user", userId: USER_ID });
182
+ const after = await decryptPiiFieldValues({ message: logs[0]?.["message"] }, ["message"], kms, {
183
+ requestId: "t",
184
+ });
185
+ expect(after["message"]).toBe(PII_ERASED_SENTINEL);
186
+ });
187
+
188
+ test("system run (no triggeredById) → completed log messages stay plaintext", async () => {
189
+ await logger.onJobStart?.("app:job:cron-sweep", "bull-c3", {});
190
+ await logger.onJobComplete?.("app:job:cron-sweep", "bull-c3", 5, [
191
+ { level: "info", message: "plain sweep log", timestamp: Temporal.Now.instant() },
192
+ ]);
193
+ const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-c3") });
194
+ expect(logs[0]?.["message"]).toBe("plain sweep log");
195
+ });
196
+
197
+ test("without a KMS the completed log message stays plaintext (rollout mode)", async () => {
198
+ await logger.onJobStart?.("app:job:export", "bull-c4", { triggeredById: USER_ID });
199
+ resetPiiSubjectKmsForTests();
200
+ await logger.onJobComplete?.("app:job:export", "bull-c4", 5, [
201
+ { level: "info", message: SECRET_LOG, timestamp: Temporal.Now.instant() },
202
+ ]);
203
+ const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-c4") });
204
+ expect(logs[0]?.["message"]).toBe(SECRET_LOG);
205
+ });
206
+
207
+ // Documents a gap rather than fixing it: if the subject's key is erased
208
+ // between onJobStart and onJobComplete, encryptLogMessages' getOrCreateDek
209
+ // throws KeyErasedError — same propagate-on-erased-key behavior as
210
+ // onJobStart's own encryptStartedPayload (no catch there either). Here the
211
+ // status updateMany already ran before the log insert throws, so the run
212
+ // is left "completed" with its log batch dropped instead of forging a
213
+ // plaintext fallback. Same wedge-state class #2246 tracks (job runs stuck
214
+ // after abnormal termination) — not something this PR fixes.
215
+ test("erase subject key BEFORE onJobComplete: log encryption throws, run status already landed as completed", async () => {
216
+ await logger.onJobStart?.("app:job:export", "bull-c5", {
217
+ triggeredById: USER_ID,
218
+ payload: SECRET_PAYLOAD,
219
+ });
220
+ await kms.eraseKey({ kind: "user", userId: USER_ID });
221
+
222
+ let threw = false;
223
+ try {
224
+ await logger.onJobComplete?.("app:job:export", "bull-c5", 5, [
225
+ { level: "info", message: SECRET_LOG, timestamp: Temporal.Now.instant() },
226
+ ]);
227
+ } catch {
228
+ threw = true;
229
+ }
230
+ expect(threw).toBe(true);
231
+
232
+ const row = await fetchOne(testDb.db, jobRunsTable, { bullJobId: "bull-c5" });
233
+ expect(row?.["status"]).toBe("completed");
234
+ const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-c5") });
235
+ expect(logs).toHaveLength(0);
236
+ });
237
+ });
238
+
239
+ // The suite above proves the write side (row is ciphertext) and decrypts by
240
+ // calling decryptPiiFieldValues directly in the test body — neither touches
241
+ // detail.query.ts's own decrypt wrapper. This block dispatches the real
242
+ // jobs:query:details HTTP handler so a regression in those eleven lines
243
+ // (wrong AAD field name, log-array key access, the row/log spread) fails a
244
+ // test instead of shipping ciphertext to the job-run detail screen.
245
+ describe("jobs:query:details decrypts log messages end-to-end (#2247)", () => {
246
+ let detailStack: TestStack;
247
+ let detailLogger: ReturnType<typeof createJobRunLogger>;
248
+
249
+ const DETAIL_USER_ID = "u-pii-detail-1";
250
+ const SECRET_DETAIL_LOG = "user export contained iban DE89370400440532013000";
251
+
252
+ beforeAll(async () => {
253
+ detailStack = await setupTestStack({
254
+ features: [createJobsFeature()],
255
+ jobs: {
256
+ consumerLane: "worker",
257
+ queueNamePrefix: `kumiko-jobs-detail-pii-test-${Date.now()}`,
258
+ },
259
+ });
260
+ await unsafePushTables(detailStack.db, { jobRunsTable, jobRunLogsTable });
261
+ detailLogger = createJobRunLogger({ db: detailStack.db, registry: detailStack.registry });
262
+ });
263
+
264
+ afterAll(async () => {
265
+ await detailStack.cleanup();
266
+ });
267
+
268
+ beforeEach(() => {
269
+ configurePiiSubjectKms(new InMemoryKmsAdapter());
270
+ });
271
+
272
+ afterEach(() => {
273
+ resetPiiSubjectKmsForTests();
274
+ });
275
+
276
+ test("jobs:query:details returns plaintext log message even though the stored row is ciphertext", async () => {
277
+ await detailLogger.onJobStart?.("app:job:export", "bull-detail-1", {
278
+ triggeredById: DETAIL_USER_ID,
279
+ });
280
+ await detailLogger.onJobComplete?.("app:job:export", "bull-detail-1", 7, [
281
+ { level: "info", message: SECRET_DETAIL_LOG, timestamp: Temporal.Now.instant() },
282
+ ]);
283
+
284
+ const row = await fetchOne(detailStack.db, jobRunsTable, { bullJobId: "bull-detail-1" });
285
+ const runId = String(row?.["id"]);
286
+ const storedLogs = await selectMany(detailStack.db, jobRunLogsTable, { runId });
287
+ expect(isPiiCiphertext(storedLogs[0]?.["message"])).toBe(true);
288
+
289
+ const result = await detailStack.http.queryOk<{
290
+ logs: readonly { message: string }[];
291
+ }>(JobQueries.details, { runId }, TestUsers.systemAdmin);
292
+
293
+ expect(result.logs).toHaveLength(1);
294
+ expect(result.logs[0]?.message).toBe(SECRET_DETAIL_LOG);
295
+ });
296
+ });
@@ -0,0 +1,148 @@
1
+ // Integration test for the stale-run-sweep job (#2246). Dispatches
2
+ // jobs:job:stale-run-sweep through the real jobRunner (setupTestStack +
3
+ // BullMQ worker), same pattern as jobs-retention.integration.test.ts — a
4
+ // hand-built JobContext would only prove the handler function works in
5
+ // isolation, not that the job is actually registered and dispatchable.
6
+
7
+ import { afterEach, describe, expect, test } from "bun:test";
8
+ import { fetchOne } from "@cosmicdrift/kumiko-framework/bun-db";
9
+ import { sql } from "@cosmicdrift/kumiko-framework/db";
10
+ import { SYSTEM_TENANT_ID } from "@cosmicdrift/kumiko-framework/engine";
11
+ import {
12
+ setupTestStack,
13
+ type TestStack,
14
+ unsafePushTables,
15
+ } from "@cosmicdrift/kumiko-framework/stack";
16
+ import { seedRow, waitFor } from "@cosmicdrift/kumiko-framework/testing";
17
+ import { STALE_JOB_RUN_ERROR } from "../db/queries/stale-run-sweep";
18
+ import { createJobsFeature } from "../feature";
19
+ import { DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS } from "../handlers/stale-run-sweep.job";
20
+ import { jobRunLogsTable, jobRunsTable } from "../job-run-table";
21
+
22
+ const SWEEP_JOB = "jobs:job:stale-run-sweep";
23
+
24
+ let stack: TestStack | undefined;
25
+
26
+ async function bootStack(staleRunTimeoutHours?: number): Promise<TestStack> {
27
+ const s = await setupTestStack({
28
+ features: [
29
+ createJobsFeature(staleRunTimeoutHours !== undefined ? { staleRunTimeoutHours } : {}),
30
+ ],
31
+ jobs: { consumerLane: "worker" },
32
+ });
33
+ await unsafePushTables(s.db, { jobRunsTable, jobRunLogsTable });
34
+ return s;
35
+ }
36
+
37
+ afterEach(async () => {
38
+ if (stack) await stack.cleanup();
39
+ stack = undefined;
40
+ });
41
+
42
+ function currentStack(): TestStack {
43
+ if (!stack) throw new Error("stack not booted");
44
+ return stack;
45
+ }
46
+
47
+ async function seedRun(opts: {
48
+ id: string;
49
+ status: "running" | "completed";
50
+ ageHours: number;
51
+ }): Promise<void> {
52
+ const startedAt = sql`now() - ${sql.raw(`interval '${opts.ageHours} hours'`)}`;
53
+ await seedRow(currentStack().db, jobRunsTable, {
54
+ id: opts.id,
55
+ tenantId: SYSTEM_TENANT_ID,
56
+ jobName: "example:job:stale-run-probe",
57
+ bullJobId: `bull-${opts.id}`,
58
+ status: opts.status,
59
+ attempt: 1,
60
+ startedAt,
61
+ });
62
+ }
63
+
64
+ async function statusOf(id: string): Promise<string | undefined> {
65
+ const row = await fetchOne<{ status: string }>(currentStack().db, jobRunsTable, { id });
66
+ return row?.status;
67
+ }
68
+
69
+ describe("jobs:job:stale-run-sweep", () => {
70
+ test("a running row older than the timeout is marked failed with the sweeper error", async () => {
71
+ stack = await bootStack();
72
+ const staleId = "11111111-1111-4111-8111-111111111111";
73
+ await seedRun({
74
+ id: staleId,
75
+ status: "running",
76
+ ageHours: DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS + 1,
77
+ });
78
+
79
+ await currentStack().jobRunner?.dispatch(SWEEP_JOB);
80
+ await waitFor(async () => {
81
+ expect(await statusOf(staleId)).toBe("failed");
82
+ });
83
+
84
+ const row = await fetchOne<{ status: string; error: string | null }>(
85
+ currentStack().db,
86
+ jobRunsTable,
87
+ { id: staleId },
88
+ );
89
+ expect(row?.error).toBe(STALE_JOB_RUN_ERROR);
90
+ });
91
+
92
+ test("a running row still within the timeout is left running, untouched", async () => {
93
+ stack = await bootStack();
94
+ const staleId = "22222222-2222-4222-8222-222222222222";
95
+ const recentId = "33333333-3333-4333-8333-333333333333";
96
+ await seedRun({
97
+ id: staleId,
98
+ status: "running",
99
+ ageHours: DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS + 1,
100
+ });
101
+ await seedRun({ id: recentId, status: "running", ageHours: 1 });
102
+
103
+ await currentStack().jobRunner?.dispatch(SWEEP_JOB);
104
+ await waitFor(async () => {
105
+ expect(await statusOf(staleId)).toBe("failed");
106
+ });
107
+
108
+ expect(await statusOf(recentId)).toBe("running");
109
+ });
110
+
111
+ test("a completed row older than the timeout is left completed, untouched", async () => {
112
+ stack = await bootStack();
113
+ const staleRunningId = "44444444-4444-4444-8444-444444444444";
114
+ const oldCompletedId = "55555555-5555-4555-8555-555555555555";
115
+ await seedRun({
116
+ id: staleRunningId,
117
+ status: "running",
118
+ ageHours: DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS + 1,
119
+ });
120
+ await seedRun({
121
+ id: oldCompletedId,
122
+ status: "completed",
123
+ ageHours: DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS + 5,
124
+ });
125
+
126
+ await currentStack().jobRunner?.dispatch(SWEEP_JOB);
127
+ await waitFor(async () => {
128
+ expect(await statusOf(staleRunningId)).toBe("failed");
129
+ });
130
+
131
+ expect(await statusOf(oldCompletedId)).toBe("completed");
132
+ });
133
+
134
+ test("a custom staleRunTimeoutHours option is honored", async () => {
135
+ stack = await bootStack(2);
136
+ const staleId = "66666666-6666-4666-8666-666666666666";
137
+ const withinCustomWindowId = "77777777-7777-4777-8777-777777777777";
138
+ await seedRun({ id: staleId, status: "running", ageHours: 3 });
139
+ await seedRun({ id: withinCustomWindowId, status: "running", ageHours: 1 });
140
+
141
+ await currentStack().jobRunner?.dispatch(SWEEP_JOB);
142
+ await waitFor(async () => {
143
+ expect(await statusOf(staleId)).toBe("failed");
144
+ });
145
+
146
+ expect(await statusOf(withinCustomWindowId)).toBe("running");
147
+ });
148
+ });
@@ -0,0 +1,51 @@
1
+ // Stale job-run sweep (#2246): store_job_runs is a direct-write store
2
+ // (#2243) — status is set to "running" once at onJobStart and only ever
3
+ // flipped to "completed"/"failed" by onJobComplete/onJobFailed. If the
4
+ // process dies mid-run (crash, OOM, kill -9), those callbacks never fire and
5
+ // the row is stuck on "running" forever — nothing else on the read side
6
+ // (list.query.ts/detail.query.ts) or in job-runner.ts (no BullMQ 'stalled'
7
+ // listener wired) ever revisits it.
8
+ //
9
+ // This sweep marks runs whose startedAt is older than the timeout as
10
+ // "failed" instead of inventing a new status: reusing "failed" means no
11
+ // schema/migration change, no web filter-dropdown/status-enum update, and
12
+ // the run becomes retriable via the existing jobs:write:retry gate
13
+ // (retry.write.ts only allows retry from "failed").
14
+
15
+ import { updateMany } from "@cosmicdrift/kumiko-framework/bun-db";
16
+ import type { DbConnection } from "@cosmicdrift/kumiko-framework/db";
17
+ import { jobRunsTable } from "../../job-run-table";
18
+
19
+ export const STALE_JOB_RUN_ERROR =
20
+ "job run exceeded the stale-run timeout without a completion signal (likely a crashed worker process)";
21
+
22
+ export type StaleRunSweepResult = {
23
+ readonly runsMarkedFailed: number;
24
+ };
25
+
26
+ export async function markStaleJobRunsFailed(
27
+ db: DbConnection,
28
+ timeoutHours: number,
29
+ ): Promise<StaleRunSweepResult> {
30
+ const cutoff = Temporal.Now.instant().subtract({ hours: timeoutHours });
31
+ const now = Temporal.Now.instant();
32
+
33
+ // duration is deliberately left untouched: we don't know when the run
34
+ // actually died, only that it crossed the timeout, so recording a
35
+ // duration would misrepresent it as measured. detail-screen/list-screen
36
+ // both already render a null duration as "—".
37
+ const updated = await updateMany(
38
+ db,
39
+ jobRunsTable,
40
+ {
41
+ status: "failed",
42
+ error: STALE_JOB_RUN_ERROR,
43
+ finishedAt: now,
44
+ modifiedAt: now,
45
+ modifiedById: "system",
46
+ },
47
+ { status: "running", startedAt: { lt: cutoff } },
48
+ );
49
+
50
+ return { runsMarkedFailed: updated.length };
51
+ }
@@ -13,6 +13,10 @@ import {
13
13
  DEFAULT_JOB_RUN_RETENTION_DAYS,
14
14
  } from "./handlers/retention-cleanup.job";
15
15
  import { retryWrite } from "./handlers/retry.write";
16
+ import {
17
+ createStaleRunSweepJob,
18
+ DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS,
19
+ } from "./handlers/stale-run-sweep.job";
16
20
  import { triggerWrite } from "./handlers/trigger.write";
17
21
  import { JOBS_I18N } from "./i18n";
18
22
  import { jobRunLogsTableMeta, jobRunsTableMeta } from "./job-run-table";
@@ -21,13 +25,20 @@ export type JobsFeatureOptions = {
21
25
  // How long a job run (and its logs) stays in store_job_runs/
22
26
  // store_job_run_logs before the daily retention-cleanup job deletes it.
23
27
  readonly retentionDays?: number;
28
+ // How long a run can sit at status "running" before the hourly
29
+ // stale-run-sweep job marks it "failed" (#2246 — a process kill mid-run
30
+ // never fires onJobComplete/onJobFailed, so nothing else ever revisits
31
+ // it). Must clear any legitimately long-running job (reindexEntity,
32
+ // projectionRebuild) — set high, not tight.
33
+ readonly staleRunTimeoutHours?: number;
24
34
  };
25
35
 
26
36
  export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefinition {
27
37
  const retentionDays = options.retentionDays ?? DEFAULT_JOB_RUN_RETENTION_DAYS;
38
+ const staleRunTimeoutHours = options.staleRunTimeoutHours ?? DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS;
28
39
  return defineFeature("jobs", (r) => {
29
40
  r.describe(
30
- "Persistence and operator tooling for background jobs registered via `r.job(...)`. Every job execution writes directly into `store_job_runs` (current status + duration) and `store_job_run_logs` (per-line log rows) from the BullMQ callbacks — no event stream in between (#2243). A daily `retention-cleanup` job deletes runs (and their logs) older than `retentionDays`. Exposes `jobs:write:trigger` (manual run) and `jobs:write:retry` (operator retry of a failed run), plus `jobs:query:list`, `jobs:query:details`, and `jobs:query:catalog` (manual jobs) for the operator UI.",
41
+ "Persistence and operator tooling for background jobs registered via `r.job(...)`. Every job execution writes directly into `store_job_runs` (current status + duration) and `store_job_run_logs` (per-line log rows) from the BullMQ callbacks — no event stream in between (#2243). A daily `retention-cleanup` job deletes runs (and their logs) older than `retentionDays`; an hourly `stale-run-sweep` job marks runs stuck at status `running` past `staleRunTimeoutHours` as `failed` (#2246 — a crashed worker never fires the completion callback, so nothing else ever revisits the row). Exposes `jobs:write:trigger` (manual run) and `jobs:write:retry` (operator retry of a failed run), plus `jobs:query:list`, `jobs:query:details`, and `jobs:query:catalog` (manual jobs) for the operator UI.",
31
42
  );
32
43
  r.uiHints({
33
44
  displayLabel: "Jobs · Audit & Operator UI",
@@ -67,6 +78,17 @@ export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefi
67
78
  createRetentionCleanupJob(retentionDays),
68
79
  );
69
80
 
81
+ // Sweeps store_job_runs for rows stuck at status "running" past the
82
+ // timeout (#2246). Hourly matches the timeout's own granularity —
83
+ // frequent enough to catch a stuck run soon after it crosses the
84
+ // threshold, cheap enough that an (almost always empty) result set
85
+ // doesn't matter.
86
+ r.job(
87
+ "stale-run-sweep",
88
+ { trigger: { cron: "0 * * * *" }, concurrency: "skip" },
89
+ createStaleRunSweepJob(staleRunTimeoutHours),
90
+ );
91
+
70
92
  const handlers = {
71
93
  trigger: r.writeHandler(triggerWrite),
72
94
  retry: r.writeHandler(retryWrite),
@@ -82,12 +104,14 @@ export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefi
82
104
 
83
105
  r.translations({ keys: JOBS_I18N });
84
106
 
107
+ // kumiko-lint-ignore app-feature-structure Phase-3 conversion tracked in #2312
85
108
  r.screen({
86
109
  id: JOB_RUNS_SCREEN_ID,
87
110
  type: "custom",
88
111
  renderer: { react: { __component: "JobRunsScreen" } },
89
112
  access: systemAdminAccess,
90
113
  });
114
+ // kumiko-lint-ignore app-feature-structure Phase-3 conversion tracked in #2312
91
115
  r.screen({
92
116
  id: JOB_RUN_DETAIL_SCREEN_ID,
93
117
  type: "custom",
@@ -39,6 +39,18 @@ export const detailQuery = defineQueryHandler({
39
39
  },
40
40
  );
41
41
 
42
- return { ...row, logs };
42
+ // message is stored encrypted under the triggering user's DEK (#2247),
43
+ // same mechanism as row.payload above.
44
+ const decryptedLogs = await Promise.all(
45
+ logs.map(async (log) => {
46
+ if (typeof log["message"] !== "string") return log;
47
+ return {
48
+ ...log,
49
+ message: await decryptStoredPii(log["message"], "message", "job-run-detail-log"),
50
+ };
51
+ }),
52
+ );
53
+
54
+ return { ...row, logs: decryptedLogs };
43
55
  },
44
56
  });
@@ -0,0 +1,28 @@
1
+ import type { DbConnection } from "@cosmicdrift/kumiko-framework/db";
2
+ import type { JobHandlerFn } from "@cosmicdrift/kumiko-framework/engine";
3
+ import { InternalError } from "@cosmicdrift/kumiko-framework/errors";
4
+ import { markStaleJobRunsFailed } from "../db/queries/stale-run-sweep";
5
+
6
+ // 24h, not e.g. the 1h cache-TTL used elsewhere in this feature (see
7
+ // job-run-logger.ts) — a wrong guess there costs one extra DB lookup, a
8
+ // wrong guess here marks a still-running job "failed" and opens
9
+ // retry.write.ts's retry gate on it while the original run is still
10
+ // executing (concurrent duplicate dispatch). reindexEntity/projectionRebuild
11
+ // (registered in this same feature) can legitimately run for hours, so the
12
+ // default has to clear any plausible real job, not just be "long".
13
+ // Configurable per-app via JobsFeatureOptions.staleRunTimeoutHours.
14
+ export const DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS = 24;
15
+
16
+ export function createStaleRunSweepJob(timeoutHours: number): JobHandlerFn {
17
+ return async (_payload, ctx) => {
18
+ if (!ctx.db) {
19
+ throw new InternalError({
20
+ message:
21
+ "[jobs:stale-run-sweep] ctx.db missing — job context requires a database connection.",
22
+ });
23
+ }
24
+ const db = ctx.db as DbConnection; // @cast-boundary db-operator (matches sibling cron jobs)
25
+ const result = await markStaleJobRunsFailed(db, timeoutHours);
26
+ ctx.log?.info?.(`[jobs:stale-run-sweep] complete: ${JSON.stringify(result)}`);
27
+ };
28
+ }