@cosmicdrift/kumiko-bundled-features 0.210.0 → 0.212.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +8 -8
- package/src/admin-shell/feature.ts +2 -0
- package/src/admin-shell/web/platform-overview-screen.tsx +1 -0
- package/src/admin-shell/web/tenant-overview-screen.tsx +1 -0
- package/src/audit/__tests__/audit-log-screen.test.tsx +228 -0
- package/src/audit/constants.ts +3 -0
- package/src/audit/feature.ts +2 -0
- package/src/audit/i18n.ts +1 -0
- package/src/audit/web/audit-log-detail-screen.tsx +28 -6
- package/src/audit/web/audit-log-screen.tsx +51 -23
- package/src/auth-email-password/feature.ts +1 -0
- package/src/auth-email-password/web/auth-form-primitives.tsx +2 -0
- package/src/auth-email-password/web/auth-gate.tsx +1 -0
- package/src/auth-email-password/web/confirm-account-unlock-screen.tsx +1 -0
- package/src/auth-email-password/web/invite-accept-screen.tsx +3 -0
- package/src/auth-email-password/web/session.tsx +1 -0
- package/src/auth-email-password/web/verify-email-screen.tsx +1 -0
- package/src/auth-mfa/feature.ts +1 -0
- package/src/auth-mfa/web/mfa-disable-dialog.tsx +1 -0
- package/src/auth-mfa/web/mfa-regenerate-recovery-dialog.tsx +1 -0
- package/src/compliance-profiles/feature.ts +1 -5
- package/src/crypto-shredding/changes.json +2 -1
- package/src/custom-fields/__tests__/feature.test.ts +42 -0
- package/src/custom-fields/__tests__/field-definition-role-config.integration.test.ts +139 -0
- package/src/custom-fields/constants.ts +6 -0
- package/src/custom-fields/feature.ts +56 -10
- package/src/custom-fields/handlers/define-tenant-field.write.ts +7 -1
- package/src/custom-fields/handlers/delete-tenant-field.write.ts +54 -33
- package/src/custom-fields/handlers/update-tenant-field.write.ts +58 -35
- package/src/delivery/feature.ts +1 -3
- package/src/feature-toggles/feature.ts +1 -0
- package/src/feature-toggles/web/toggle-admin-screen.tsx +1 -0
- package/src/file-derivatives/changes.json +9 -1
- package/src/folders/web/folder-manager.tsx +1 -0
- package/src/jobs/__tests__/jobs-pii-kms.integration.test.ts +174 -0
- package/src/jobs/__tests__/jobs-stale-run-sweep.integration.test.ts +148 -0
- package/src/jobs/db/queries/stale-run-sweep.ts +51 -0
- package/src/jobs/feature.ts +25 -1
- package/src/jobs/handlers/detail.query.ts +13 -1
- package/src/jobs/handlers/stale-run-sweep.job.ts +28 -0
- package/src/jobs/job-run-logger.ts +68 -18
- package/src/jobs/web/job-run-detail-screen.tsx +1 -0
- package/src/page-render/web.ts +1 -0
- package/src/personal-access-tokens/changes.json +7 -0
- package/src/personal-access-tokens/feature.ts +1 -0
- package/src/secrets/__tests__/access-options.integration.test.ts +114 -0
- package/src/secrets/constants.ts +12 -0
- package/src/secrets/feature.ts +35 -9
- package/src/secrets/handlers/delete.write.ts +23 -16
- package/src/secrets/handlers/list.query.ts +37 -30
- package/src/secrets/handlers/set.write.ts +43 -36
- package/src/secrets/index.ts +5 -0
- package/src/seo/feature.ts +1 -0
- package/src/tags/feature.ts +1 -0
- package/src/tags/web/tag-chip.tsx +2 -0
- package/src/tags/web/tag-manager.tsx +2 -1
- package/src/tags/web/tag-picker.tsx +1 -0
- package/src/template-resolver/changes.json +9 -1
- package/src/template-resolver/web/client-plugin.tsx +2 -0
- package/src/tenant/__tests__/members-screens.boot.test.ts +79 -4
- package/src/tenant/__tests__/tenant-security.integration.test.ts +260 -3
- package/src/tenant/changes.json +6 -0
- package/src/tenant/constants.ts +17 -4
- package/src/tenant/feature.ts +28 -11
- package/src/tenant/handlers/team-list.query.ts +250 -0
- package/src/tenant/i18n.ts +20 -0
- package/src/tenant/screens.ts +111 -3
- package/src/tenant/web/client-plugin.tsx +11 -6
- package/src/tenant/web/index.ts +2 -2
- package/src/tenant/web/member-roles-cell.tsx +12 -0
- package/src/tenant/web/member-status-cell.tsx +23 -0
- package/src/tier-engine/feature.ts +2 -0
- package/src/tier-engine/web/tier-admin-screen.tsx +5 -5
- package/src/user-data-rights/feature.ts +2 -0
- package/src/user-data-rights/web/privacy-center-screen.tsx +2 -1
- package/src/user-profile/i18n.ts +1 -0
- package/src/user-profile/web/profile-screen.tsx +1 -1
- package/src/tenant/web/i18n.ts +0 -24
- package/src/tenant/web/members-screen.tsx +0 -270
|
@@ -1 +1,9 @@
|
|
|
1
|
-
[
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"version": "0.194.0",
|
|
4
|
+
"type": "breaking",
|
|
5
|
+
"title": "PRESET_VARIANT_NAMES removed from public exports; publicVariantQuery accepts any name.",
|
|
6
|
+
"detail": "The public GET /media/:fileRefId/:variant route now resolves the variant spec from the FileRef's field declaration (createImageField({ variants: {...} })) instead of a fixed spec table, so an app can declare and publicly serve its own variant names, and a field overriding a built-in preset name (thumb/card/hero/full) is served with its own spec instead of the frozen preset size. publicVariantQuery's schema now accepts any variant name (z.string().min(1).max(64)) instead of z.enum(PRESET_VARIANT_NAMES); resolution requires an exact match against the field's declared variants keys, and an unresolvable name answers 404, same as an unknown FileRef. The route's pre-DB path-param gate is now purely syntactic ([a-zA-Z0-9_-]{1,64}) instead of a name allow-list; a known-valid preset name plus a random fileRefId reached the same DB read as any other name, so the list only blocked the cheaper of two equally-costly attacks, and the actual defense is the existing per-IP rate limit and the UUID guard, both unchanged.",
|
|
7
|
+
"migration": "PRESET_VARIANT_NAMES only ever named the allow-list this release deletes, so there is nothing to migrate to. thumb/card/hero/full remain as ready-made specs to spread into a field's own variants."
|
|
8
|
+
}
|
|
9
|
+
]
|
|
@@ -108,6 +108,7 @@ export function FolderManager({
|
|
|
108
108
|
const [pendingDelete, setPendingDelete] = useState<FolderNode | null>(null);
|
|
109
109
|
|
|
110
110
|
// Persist expand/collapse so a reload (F5) doesn't re-expand everything.
|
|
111
|
+
// kumiko-lint-ignore no-raw-hooks DOM integration: writes to window.localStorage, no framework query/mutation equivalent
|
|
111
112
|
useEffect(() => {
|
|
112
113
|
if (typeof window === "undefined") return;
|
|
113
114
|
try {
|
|
@@ -19,11 +19,15 @@ import { createEventsTable, eventsTable } from "@cosmicdrift/kumiko-framework/ev
|
|
|
19
19
|
import {
|
|
20
20
|
createTestDb,
|
|
21
21
|
createTestRedis,
|
|
22
|
+
setupTestStack,
|
|
22
23
|
type TestDb,
|
|
23
24
|
type TestRedis,
|
|
25
|
+
type TestStack,
|
|
26
|
+
TestUsers,
|
|
24
27
|
unsafePushTables,
|
|
25
28
|
} from "@cosmicdrift/kumiko-framework/stack";
|
|
26
29
|
import { resetPiiSubjectKmsForTests, resetTestTables } from "@cosmicdrift/kumiko-framework/testing";
|
|
30
|
+
import { JobQueries } from "../constants";
|
|
27
31
|
import { createJobsFeature } from "../feature";
|
|
28
32
|
import { createJobRunLogger } from "../job-run-logger";
|
|
29
33
|
import { jobRunLogsTable, jobRunsTable } from "../job-run-table";
|
|
@@ -120,3 +124,173 @@ describe("jobs run-started payload under KMS", () => {
|
|
|
120
124
|
expect(row?.["payload"]).toBe(SECRET_PAYLOAD);
|
|
121
125
|
});
|
|
122
126
|
});
|
|
127
|
+
|
|
128
|
+
// #2247: store_job_run_logs is the same unmanaged direct-write path as
|
|
129
|
+
// jobRunsTable.payload above — onJobComplete/onJobFailed batch-insert log
|
|
130
|
+
// lines straight from the BullMQ callback, with no event-PII catalog to
|
|
131
|
+
// lean on. Same subject (triggeredById), same encrypt-under-DEK treatment,
|
|
132
|
+
// same skip rules (null subject / no KMS stay plaintext).
|
|
133
|
+
describe("jobs run-completed/-failed log messages under KMS (#2247)", () => {
|
|
134
|
+
const SECRET_LOG = "user export contained iban DE89370400440532013000";
|
|
135
|
+
|
|
136
|
+
async function runIdFor(bullJobId: string): Promise<string> {
|
|
137
|
+
const row = await fetchOne(testDb.db, jobRunsTable, { bullJobId });
|
|
138
|
+
return String(row?.["id"]);
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
test("onJobComplete: stored log message is ciphertext, decrypts to original", async () => {
|
|
142
|
+
await logger.onJobStart?.("app:job:export", "bull-c1", { triggeredById: USER_ID });
|
|
143
|
+
await logger.onJobComplete?.("app:job:export", "bull-c1", 42, [
|
|
144
|
+
{ level: "info", message: SECRET_LOG, timestamp: Temporal.Now.instant() },
|
|
145
|
+
]);
|
|
146
|
+
|
|
147
|
+
const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-c1") });
|
|
148
|
+
expect(logs).toHaveLength(1);
|
|
149
|
+
expect(isPiiCiphertext(logs[0]?.["message"])).toBe(true);
|
|
150
|
+
expect(String(logs[0]?.["message"])).toContain(`user:${USER_ID}`);
|
|
151
|
+
|
|
152
|
+
const back = await decryptPiiFieldValues({ message: logs[0]?.["message"] }, ["message"], kms, {
|
|
153
|
+
requestId: "t",
|
|
154
|
+
});
|
|
155
|
+
expect(back["message"]).toBe(SECRET_LOG);
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
test("onJobFailed: stored log message is ciphertext, decrypts to original", async () => {
|
|
159
|
+
await logger.onJobStart?.("app:job:export", "bull-f1", { triggeredById: USER_ID });
|
|
160
|
+
await logger.onJobFailed?.("app:job:export", "bull-f1", "boom", [
|
|
161
|
+
{ level: "error", message: SECRET_LOG, timestamp: Temporal.Now.instant() },
|
|
162
|
+
]);
|
|
163
|
+
|
|
164
|
+
const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-f1") });
|
|
165
|
+
expect(logs).toHaveLength(1);
|
|
166
|
+
expect(isPiiCiphertext(logs[0]?.["message"])).toBe(true);
|
|
167
|
+
|
|
168
|
+
const back = await decryptPiiFieldValues({ message: logs[0]?.["message"] }, ["message"], kms, {
|
|
169
|
+
requestId: "t",
|
|
170
|
+
});
|
|
171
|
+
expect(back["message"]).toBe(SECRET_LOG);
|
|
172
|
+
});
|
|
173
|
+
|
|
174
|
+
test("erase subject key after write → completed log message decrypts to [[erased]]", async () => {
|
|
175
|
+
await logger.onJobStart?.("app:job:export", "bull-c2", { triggeredById: USER_ID });
|
|
176
|
+
await logger.onJobComplete?.("app:job:export", "bull-c2", 10, [
|
|
177
|
+
{ level: "info", message: SECRET_LOG, timestamp: Temporal.Now.instant() },
|
|
178
|
+
]);
|
|
179
|
+
const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-c2") });
|
|
180
|
+
|
|
181
|
+
await kms.eraseKey({ kind: "user", userId: USER_ID });
|
|
182
|
+
const after = await decryptPiiFieldValues({ message: logs[0]?.["message"] }, ["message"], kms, {
|
|
183
|
+
requestId: "t",
|
|
184
|
+
});
|
|
185
|
+
expect(after["message"]).toBe(PII_ERASED_SENTINEL);
|
|
186
|
+
});
|
|
187
|
+
|
|
188
|
+
test("system run (no triggeredById) → completed log messages stay plaintext", async () => {
|
|
189
|
+
await logger.onJobStart?.("app:job:cron-sweep", "bull-c3", {});
|
|
190
|
+
await logger.onJobComplete?.("app:job:cron-sweep", "bull-c3", 5, [
|
|
191
|
+
{ level: "info", message: "plain sweep log", timestamp: Temporal.Now.instant() },
|
|
192
|
+
]);
|
|
193
|
+
const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-c3") });
|
|
194
|
+
expect(logs[0]?.["message"]).toBe("plain sweep log");
|
|
195
|
+
});
|
|
196
|
+
|
|
197
|
+
test("without a KMS the completed log message stays plaintext (rollout mode)", async () => {
|
|
198
|
+
await logger.onJobStart?.("app:job:export", "bull-c4", { triggeredById: USER_ID });
|
|
199
|
+
resetPiiSubjectKmsForTests();
|
|
200
|
+
await logger.onJobComplete?.("app:job:export", "bull-c4", 5, [
|
|
201
|
+
{ level: "info", message: SECRET_LOG, timestamp: Temporal.Now.instant() },
|
|
202
|
+
]);
|
|
203
|
+
const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-c4") });
|
|
204
|
+
expect(logs[0]?.["message"]).toBe(SECRET_LOG);
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
// Documents a gap rather than fixing it: if the subject's key is erased
|
|
208
|
+
// between onJobStart and onJobComplete, encryptLogMessages' getOrCreateDek
|
|
209
|
+
// throws KeyErasedError — same propagate-on-erased-key behavior as
|
|
210
|
+
// onJobStart's own encryptStartedPayload (no catch there either). Here the
|
|
211
|
+
// status updateMany already ran before the log insert throws, so the run
|
|
212
|
+
// is left "completed" with its log batch dropped instead of forging a
|
|
213
|
+
// plaintext fallback. Same wedge-state class #2246 tracks (job runs stuck
|
|
214
|
+
// after abnormal termination) — not something this PR fixes.
|
|
215
|
+
test("erase subject key BEFORE onJobComplete: log encryption throws, run status already landed as completed", async () => {
|
|
216
|
+
await logger.onJobStart?.("app:job:export", "bull-c5", {
|
|
217
|
+
triggeredById: USER_ID,
|
|
218
|
+
payload: SECRET_PAYLOAD,
|
|
219
|
+
});
|
|
220
|
+
await kms.eraseKey({ kind: "user", userId: USER_ID });
|
|
221
|
+
|
|
222
|
+
let threw = false;
|
|
223
|
+
try {
|
|
224
|
+
await logger.onJobComplete?.("app:job:export", "bull-c5", 5, [
|
|
225
|
+
{ level: "info", message: SECRET_LOG, timestamp: Temporal.Now.instant() },
|
|
226
|
+
]);
|
|
227
|
+
} catch {
|
|
228
|
+
threw = true;
|
|
229
|
+
}
|
|
230
|
+
expect(threw).toBe(true);
|
|
231
|
+
|
|
232
|
+
const row = await fetchOne(testDb.db, jobRunsTable, { bullJobId: "bull-c5" });
|
|
233
|
+
expect(row?.["status"]).toBe("completed");
|
|
234
|
+
const logs = await selectMany(testDb.db, jobRunLogsTable, { runId: await runIdFor("bull-c5") });
|
|
235
|
+
expect(logs).toHaveLength(0);
|
|
236
|
+
});
|
|
237
|
+
});
|
|
238
|
+
|
|
239
|
+
// The suite above proves the write side (row is ciphertext) and decrypts by
|
|
240
|
+
// calling decryptPiiFieldValues directly in the test body — neither touches
|
|
241
|
+
// detail.query.ts's own decrypt wrapper. This block dispatches the real
|
|
242
|
+
// jobs:query:details HTTP handler so a regression in those eleven lines
|
|
243
|
+
// (wrong AAD field name, log-array key access, the row/log spread) fails a
|
|
244
|
+
// test instead of shipping ciphertext to the job-run detail screen.
|
|
245
|
+
describe("jobs:query:details decrypts log messages end-to-end (#2247)", () => {
|
|
246
|
+
let detailStack: TestStack;
|
|
247
|
+
let detailLogger: ReturnType<typeof createJobRunLogger>;
|
|
248
|
+
|
|
249
|
+
const DETAIL_USER_ID = "u-pii-detail-1";
|
|
250
|
+
const SECRET_DETAIL_LOG = "user export contained iban DE89370400440532013000";
|
|
251
|
+
|
|
252
|
+
beforeAll(async () => {
|
|
253
|
+
detailStack = await setupTestStack({
|
|
254
|
+
features: [createJobsFeature()],
|
|
255
|
+
jobs: {
|
|
256
|
+
consumerLane: "worker",
|
|
257
|
+
queueNamePrefix: `kumiko-jobs-detail-pii-test-${Date.now()}`,
|
|
258
|
+
},
|
|
259
|
+
});
|
|
260
|
+
await unsafePushTables(detailStack.db, { jobRunsTable, jobRunLogsTable });
|
|
261
|
+
detailLogger = createJobRunLogger({ db: detailStack.db, registry: detailStack.registry });
|
|
262
|
+
});
|
|
263
|
+
|
|
264
|
+
afterAll(async () => {
|
|
265
|
+
await detailStack.cleanup();
|
|
266
|
+
});
|
|
267
|
+
|
|
268
|
+
beforeEach(() => {
|
|
269
|
+
configurePiiSubjectKms(new InMemoryKmsAdapter());
|
|
270
|
+
});
|
|
271
|
+
|
|
272
|
+
afterEach(() => {
|
|
273
|
+
resetPiiSubjectKmsForTests();
|
|
274
|
+
});
|
|
275
|
+
|
|
276
|
+
test("jobs:query:details returns plaintext log message even though the stored row is ciphertext", async () => {
|
|
277
|
+
await detailLogger.onJobStart?.("app:job:export", "bull-detail-1", {
|
|
278
|
+
triggeredById: DETAIL_USER_ID,
|
|
279
|
+
});
|
|
280
|
+
await detailLogger.onJobComplete?.("app:job:export", "bull-detail-1", 7, [
|
|
281
|
+
{ level: "info", message: SECRET_DETAIL_LOG, timestamp: Temporal.Now.instant() },
|
|
282
|
+
]);
|
|
283
|
+
|
|
284
|
+
const row = await fetchOne(detailStack.db, jobRunsTable, { bullJobId: "bull-detail-1" });
|
|
285
|
+
const runId = String(row?.["id"]);
|
|
286
|
+
const storedLogs = await selectMany(detailStack.db, jobRunLogsTable, { runId });
|
|
287
|
+
expect(isPiiCiphertext(storedLogs[0]?.["message"])).toBe(true);
|
|
288
|
+
|
|
289
|
+
const result = await detailStack.http.queryOk<{
|
|
290
|
+
logs: readonly { message: string }[];
|
|
291
|
+
}>(JobQueries.details, { runId }, TestUsers.systemAdmin);
|
|
292
|
+
|
|
293
|
+
expect(result.logs).toHaveLength(1);
|
|
294
|
+
expect(result.logs[0]?.message).toBe(SECRET_DETAIL_LOG);
|
|
295
|
+
});
|
|
296
|
+
});
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
// Integration test for the stale-run-sweep job (#2246). Dispatches
|
|
2
|
+
// jobs:job:stale-run-sweep through the real jobRunner (setupTestStack +
|
|
3
|
+
// BullMQ worker), same pattern as jobs-retention.integration.test.ts — a
|
|
4
|
+
// hand-built JobContext would only prove the handler function works in
|
|
5
|
+
// isolation, not that the job is actually registered and dispatchable.
|
|
6
|
+
|
|
7
|
+
import { afterEach, describe, expect, test } from "bun:test";
|
|
8
|
+
import { fetchOne } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
9
|
+
import { sql } from "@cosmicdrift/kumiko-framework/db";
|
|
10
|
+
import { SYSTEM_TENANT_ID } from "@cosmicdrift/kumiko-framework/engine";
|
|
11
|
+
import {
|
|
12
|
+
setupTestStack,
|
|
13
|
+
type TestStack,
|
|
14
|
+
unsafePushTables,
|
|
15
|
+
} from "@cosmicdrift/kumiko-framework/stack";
|
|
16
|
+
import { seedRow, waitFor } from "@cosmicdrift/kumiko-framework/testing";
|
|
17
|
+
import { STALE_JOB_RUN_ERROR } from "../db/queries/stale-run-sweep";
|
|
18
|
+
import { createJobsFeature } from "../feature";
|
|
19
|
+
import { DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS } from "../handlers/stale-run-sweep.job";
|
|
20
|
+
import { jobRunLogsTable, jobRunsTable } from "../job-run-table";
|
|
21
|
+
|
|
22
|
+
const SWEEP_JOB = "jobs:job:stale-run-sweep";
|
|
23
|
+
|
|
24
|
+
let stack: TestStack | undefined;
|
|
25
|
+
|
|
26
|
+
async function bootStack(staleRunTimeoutHours?: number): Promise<TestStack> {
|
|
27
|
+
const s = await setupTestStack({
|
|
28
|
+
features: [
|
|
29
|
+
createJobsFeature(staleRunTimeoutHours !== undefined ? { staleRunTimeoutHours } : {}),
|
|
30
|
+
],
|
|
31
|
+
jobs: { consumerLane: "worker" },
|
|
32
|
+
});
|
|
33
|
+
await unsafePushTables(s.db, { jobRunsTable, jobRunLogsTable });
|
|
34
|
+
return s;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
afterEach(async () => {
|
|
38
|
+
if (stack) await stack.cleanup();
|
|
39
|
+
stack = undefined;
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
function currentStack(): TestStack {
|
|
43
|
+
if (!stack) throw new Error("stack not booted");
|
|
44
|
+
return stack;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
async function seedRun(opts: {
|
|
48
|
+
id: string;
|
|
49
|
+
status: "running" | "completed";
|
|
50
|
+
ageHours: number;
|
|
51
|
+
}): Promise<void> {
|
|
52
|
+
const startedAt = sql`now() - ${sql.raw(`interval '${opts.ageHours} hours'`)}`;
|
|
53
|
+
await seedRow(currentStack().db, jobRunsTable, {
|
|
54
|
+
id: opts.id,
|
|
55
|
+
tenantId: SYSTEM_TENANT_ID,
|
|
56
|
+
jobName: "example:job:stale-run-probe",
|
|
57
|
+
bullJobId: `bull-${opts.id}`,
|
|
58
|
+
status: opts.status,
|
|
59
|
+
attempt: 1,
|
|
60
|
+
startedAt,
|
|
61
|
+
});
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
async function statusOf(id: string): Promise<string | undefined> {
|
|
65
|
+
const row = await fetchOne<{ status: string }>(currentStack().db, jobRunsTable, { id });
|
|
66
|
+
return row?.status;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
describe("jobs:job:stale-run-sweep", () => {
|
|
70
|
+
test("a running row older than the timeout is marked failed with the sweeper error", async () => {
|
|
71
|
+
stack = await bootStack();
|
|
72
|
+
const staleId = "11111111-1111-4111-8111-111111111111";
|
|
73
|
+
await seedRun({
|
|
74
|
+
id: staleId,
|
|
75
|
+
status: "running",
|
|
76
|
+
ageHours: DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS + 1,
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
await currentStack().jobRunner?.dispatch(SWEEP_JOB);
|
|
80
|
+
await waitFor(async () => {
|
|
81
|
+
expect(await statusOf(staleId)).toBe("failed");
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
const row = await fetchOne<{ status: string; error: string | null }>(
|
|
85
|
+
currentStack().db,
|
|
86
|
+
jobRunsTable,
|
|
87
|
+
{ id: staleId },
|
|
88
|
+
);
|
|
89
|
+
expect(row?.error).toBe(STALE_JOB_RUN_ERROR);
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
test("a running row still within the timeout is left running, untouched", async () => {
|
|
93
|
+
stack = await bootStack();
|
|
94
|
+
const staleId = "22222222-2222-4222-8222-222222222222";
|
|
95
|
+
const recentId = "33333333-3333-4333-8333-333333333333";
|
|
96
|
+
await seedRun({
|
|
97
|
+
id: staleId,
|
|
98
|
+
status: "running",
|
|
99
|
+
ageHours: DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS + 1,
|
|
100
|
+
});
|
|
101
|
+
await seedRun({ id: recentId, status: "running", ageHours: 1 });
|
|
102
|
+
|
|
103
|
+
await currentStack().jobRunner?.dispatch(SWEEP_JOB);
|
|
104
|
+
await waitFor(async () => {
|
|
105
|
+
expect(await statusOf(staleId)).toBe("failed");
|
|
106
|
+
});
|
|
107
|
+
|
|
108
|
+
expect(await statusOf(recentId)).toBe("running");
|
|
109
|
+
});
|
|
110
|
+
|
|
111
|
+
test("a completed row older than the timeout is left completed, untouched", async () => {
|
|
112
|
+
stack = await bootStack();
|
|
113
|
+
const staleRunningId = "44444444-4444-4444-8444-444444444444";
|
|
114
|
+
const oldCompletedId = "55555555-5555-4555-8555-555555555555";
|
|
115
|
+
await seedRun({
|
|
116
|
+
id: staleRunningId,
|
|
117
|
+
status: "running",
|
|
118
|
+
ageHours: DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS + 1,
|
|
119
|
+
});
|
|
120
|
+
await seedRun({
|
|
121
|
+
id: oldCompletedId,
|
|
122
|
+
status: "completed",
|
|
123
|
+
ageHours: DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS + 5,
|
|
124
|
+
});
|
|
125
|
+
|
|
126
|
+
await currentStack().jobRunner?.dispatch(SWEEP_JOB);
|
|
127
|
+
await waitFor(async () => {
|
|
128
|
+
expect(await statusOf(staleRunningId)).toBe("failed");
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
expect(await statusOf(oldCompletedId)).toBe("completed");
|
|
132
|
+
});
|
|
133
|
+
|
|
134
|
+
test("a custom staleRunTimeoutHours option is honored", async () => {
|
|
135
|
+
stack = await bootStack(2);
|
|
136
|
+
const staleId = "66666666-6666-4666-8666-666666666666";
|
|
137
|
+
const withinCustomWindowId = "77777777-7777-4777-8777-777777777777";
|
|
138
|
+
await seedRun({ id: staleId, status: "running", ageHours: 3 });
|
|
139
|
+
await seedRun({ id: withinCustomWindowId, status: "running", ageHours: 1 });
|
|
140
|
+
|
|
141
|
+
await currentStack().jobRunner?.dispatch(SWEEP_JOB);
|
|
142
|
+
await waitFor(async () => {
|
|
143
|
+
expect(await statusOf(staleId)).toBe("failed");
|
|
144
|
+
});
|
|
145
|
+
|
|
146
|
+
expect(await statusOf(withinCustomWindowId)).toBe("running");
|
|
147
|
+
});
|
|
148
|
+
});
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
// Stale job-run sweep (#2246): store_job_runs is a direct-write store
|
|
2
|
+
// (#2243) — status is set to "running" once at onJobStart and only ever
|
|
3
|
+
// flipped to "completed"/"failed" by onJobComplete/onJobFailed. If the
|
|
4
|
+
// process dies mid-run (crash, OOM, kill -9), those callbacks never fire and
|
|
5
|
+
// the row is stuck on "running" forever — nothing else on the read side
|
|
6
|
+
// (list.query.ts/detail.query.ts) or in job-runner.ts (no BullMQ 'stalled'
|
|
7
|
+
// listener wired) ever revisits it.
|
|
8
|
+
//
|
|
9
|
+
// This sweep marks runs whose startedAt is older than the timeout as
|
|
10
|
+
// "failed" instead of inventing a new status: reusing "failed" means no
|
|
11
|
+
// schema/migration change, no web filter-dropdown/status-enum update, and
|
|
12
|
+
// the run becomes retriable via the existing jobs:write:retry gate
|
|
13
|
+
// (retry.write.ts only allows retry from "failed").
|
|
14
|
+
|
|
15
|
+
import { updateMany } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
16
|
+
import type { DbConnection } from "@cosmicdrift/kumiko-framework/db";
|
|
17
|
+
import { jobRunsTable } from "../../job-run-table";
|
|
18
|
+
|
|
19
|
+
export const STALE_JOB_RUN_ERROR =
|
|
20
|
+
"job run exceeded the stale-run timeout without a completion signal (likely a crashed worker process)";
|
|
21
|
+
|
|
22
|
+
export type StaleRunSweepResult = {
|
|
23
|
+
readonly runsMarkedFailed: number;
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
export async function markStaleJobRunsFailed(
|
|
27
|
+
db: DbConnection,
|
|
28
|
+
timeoutHours: number,
|
|
29
|
+
): Promise<StaleRunSweepResult> {
|
|
30
|
+
const cutoff = Temporal.Now.instant().subtract({ hours: timeoutHours });
|
|
31
|
+
const now = Temporal.Now.instant();
|
|
32
|
+
|
|
33
|
+
// duration is deliberately left untouched: we don't know when the run
|
|
34
|
+
// actually died, only that it crossed the timeout, so recording a
|
|
35
|
+
// duration would misrepresent it as measured. detail-screen/list-screen
|
|
36
|
+
// both already render a null duration as "—".
|
|
37
|
+
const updated = await updateMany(
|
|
38
|
+
db,
|
|
39
|
+
jobRunsTable,
|
|
40
|
+
{
|
|
41
|
+
status: "failed",
|
|
42
|
+
error: STALE_JOB_RUN_ERROR,
|
|
43
|
+
finishedAt: now,
|
|
44
|
+
modifiedAt: now,
|
|
45
|
+
modifiedById: "system",
|
|
46
|
+
},
|
|
47
|
+
{ status: "running", startedAt: { lt: cutoff } },
|
|
48
|
+
);
|
|
49
|
+
|
|
50
|
+
return { runsMarkedFailed: updated.length };
|
|
51
|
+
}
|
package/src/jobs/feature.ts
CHANGED
|
@@ -13,6 +13,10 @@ import {
|
|
|
13
13
|
DEFAULT_JOB_RUN_RETENTION_DAYS,
|
|
14
14
|
} from "./handlers/retention-cleanup.job";
|
|
15
15
|
import { retryWrite } from "./handlers/retry.write";
|
|
16
|
+
import {
|
|
17
|
+
createStaleRunSweepJob,
|
|
18
|
+
DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS,
|
|
19
|
+
} from "./handlers/stale-run-sweep.job";
|
|
16
20
|
import { triggerWrite } from "./handlers/trigger.write";
|
|
17
21
|
import { JOBS_I18N } from "./i18n";
|
|
18
22
|
import { jobRunLogsTableMeta, jobRunsTableMeta } from "./job-run-table";
|
|
@@ -21,13 +25,20 @@ export type JobsFeatureOptions = {
|
|
|
21
25
|
// How long a job run (and its logs) stays in store_job_runs/
|
|
22
26
|
// store_job_run_logs before the daily retention-cleanup job deletes it.
|
|
23
27
|
readonly retentionDays?: number;
|
|
28
|
+
// How long a run can sit at status "running" before the hourly
|
|
29
|
+
// stale-run-sweep job marks it "failed" (#2246 — a process kill mid-run
|
|
30
|
+
// never fires onJobComplete/onJobFailed, so nothing else ever revisits
|
|
31
|
+
// it). Must clear any legitimately long-running job (reindexEntity,
|
|
32
|
+
// projectionRebuild) — set high, not tight.
|
|
33
|
+
readonly staleRunTimeoutHours?: number;
|
|
24
34
|
};
|
|
25
35
|
|
|
26
36
|
export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefinition {
|
|
27
37
|
const retentionDays = options.retentionDays ?? DEFAULT_JOB_RUN_RETENTION_DAYS;
|
|
38
|
+
const staleRunTimeoutHours = options.staleRunTimeoutHours ?? DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS;
|
|
28
39
|
return defineFeature("jobs", (r) => {
|
|
29
40
|
r.describe(
|
|
30
|
-
"Persistence and operator tooling for background jobs registered via `r.job(...)`. Every job execution writes directly into `store_job_runs` (current status + duration) and `store_job_run_logs` (per-line log rows) from the BullMQ callbacks — no event stream in between (#2243). A daily `retention-cleanup` job deletes runs (and their logs) older than `retentionDays
|
|
41
|
+
"Persistence and operator tooling for background jobs registered via `r.job(...)`. Every job execution writes directly into `store_job_runs` (current status + duration) and `store_job_run_logs` (per-line log rows) from the BullMQ callbacks — no event stream in between (#2243). A daily `retention-cleanup` job deletes runs (and their logs) older than `retentionDays`; an hourly `stale-run-sweep` job marks runs stuck at status `running` past `staleRunTimeoutHours` as `failed` (#2246 — a crashed worker never fires the completion callback, so nothing else ever revisits the row). Exposes `jobs:write:trigger` (manual run) and `jobs:write:retry` (operator retry of a failed run), plus `jobs:query:list`, `jobs:query:details`, and `jobs:query:catalog` (manual jobs) for the operator UI.",
|
|
31
42
|
);
|
|
32
43
|
r.uiHints({
|
|
33
44
|
displayLabel: "Jobs · Audit & Operator UI",
|
|
@@ -67,6 +78,17 @@ export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefi
|
|
|
67
78
|
createRetentionCleanupJob(retentionDays),
|
|
68
79
|
);
|
|
69
80
|
|
|
81
|
+
// Sweeps store_job_runs for rows stuck at status "running" past the
|
|
82
|
+
// timeout (#2246). Hourly matches the timeout's own granularity —
|
|
83
|
+
// frequent enough to catch a stuck run soon after it crosses the
|
|
84
|
+
// threshold, cheap enough that an (almost always empty) result set
|
|
85
|
+
// doesn't matter.
|
|
86
|
+
r.job(
|
|
87
|
+
"stale-run-sweep",
|
|
88
|
+
{ trigger: { cron: "0 * * * *" }, concurrency: "skip" },
|
|
89
|
+
createStaleRunSweepJob(staleRunTimeoutHours),
|
|
90
|
+
);
|
|
91
|
+
|
|
70
92
|
const handlers = {
|
|
71
93
|
trigger: r.writeHandler(triggerWrite),
|
|
72
94
|
retry: r.writeHandler(retryWrite),
|
|
@@ -82,12 +104,14 @@ export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefi
|
|
|
82
104
|
|
|
83
105
|
r.translations({ keys: JOBS_I18N });
|
|
84
106
|
|
|
107
|
+
// kumiko-lint-ignore app-feature-structure Phase-3 conversion tracked in #2312
|
|
85
108
|
r.screen({
|
|
86
109
|
id: JOB_RUNS_SCREEN_ID,
|
|
87
110
|
type: "custom",
|
|
88
111
|
renderer: { react: { __component: "JobRunsScreen" } },
|
|
89
112
|
access: systemAdminAccess,
|
|
90
113
|
});
|
|
114
|
+
// kumiko-lint-ignore app-feature-structure Phase-3 conversion tracked in #2312
|
|
91
115
|
r.screen({
|
|
92
116
|
id: JOB_RUN_DETAIL_SCREEN_ID,
|
|
93
117
|
type: "custom",
|
|
@@ -39,6 +39,18 @@ export const detailQuery = defineQueryHandler({
|
|
|
39
39
|
},
|
|
40
40
|
);
|
|
41
41
|
|
|
42
|
-
|
|
42
|
+
// message is stored encrypted under the triggering user's DEK (#2247),
|
|
43
|
+
// same mechanism as row.payload above.
|
|
44
|
+
const decryptedLogs = await Promise.all(
|
|
45
|
+
logs.map(async (log) => {
|
|
46
|
+
if (typeof log["message"] !== "string") return log;
|
|
47
|
+
return {
|
|
48
|
+
...log,
|
|
49
|
+
message: await decryptStoredPii(log["message"], "message", "job-run-detail-log"),
|
|
50
|
+
};
|
|
51
|
+
}),
|
|
52
|
+
);
|
|
53
|
+
|
|
54
|
+
return { ...row, logs: decryptedLogs };
|
|
43
55
|
},
|
|
44
56
|
});
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import type { DbConnection } from "@cosmicdrift/kumiko-framework/db";
|
|
2
|
+
import type { JobHandlerFn } from "@cosmicdrift/kumiko-framework/engine";
|
|
3
|
+
import { InternalError } from "@cosmicdrift/kumiko-framework/errors";
|
|
4
|
+
import { markStaleJobRunsFailed } from "../db/queries/stale-run-sweep";
|
|
5
|
+
|
|
6
|
+
// 24h, not e.g. the 1h cache-TTL used elsewhere in this feature (see
|
|
7
|
+
// job-run-logger.ts) — a wrong guess there costs one extra DB lookup, a
|
|
8
|
+
// wrong guess here marks a still-running job "failed" and opens
|
|
9
|
+
// retry.write.ts's retry gate on it while the original run is still
|
|
10
|
+
// executing (concurrent duplicate dispatch). reindexEntity/projectionRebuild
|
|
11
|
+
// (registered in this same feature) can legitimately run for hours, so the
|
|
12
|
+
// default has to clear any plausible real job, not just be "long".
|
|
13
|
+
// Configurable per-app via JobsFeatureOptions.staleRunTimeoutHours.
|
|
14
|
+
export const DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS = 24;
|
|
15
|
+
|
|
16
|
+
export function createStaleRunSweepJob(timeoutHours: number): JobHandlerFn {
|
|
17
|
+
return async (_payload, ctx) => {
|
|
18
|
+
if (!ctx.db) {
|
|
19
|
+
throw new InternalError({
|
|
20
|
+
message:
|
|
21
|
+
"[jobs:stale-run-sweep] ctx.db missing — job context requires a database connection.",
|
|
22
|
+
});
|
|
23
|
+
}
|
|
24
|
+
const db = ctx.db as DbConnection; // @cast-boundary db-operator (matches sibling cron jobs)
|
|
25
|
+
const result = await markStaleJobRunsFailed(db, timeoutHours);
|
|
26
|
+
ctx.log?.info?.(`[jobs:stale-run-sweep] complete: ${JSON.stringify(result)}`);
|
|
27
|
+
};
|
|
28
|
+
}
|