@cosmicdrift/kumiko-bundled-features 0.289.0 → 0.291.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +9 -9
- package/src/auth-email-password/__tests__/query-as-member.integration.test.ts +15 -2
- package/src/auth-email-password/changes.json +6 -0
- package/src/auth-email-password/web/__tests__/auth-form-logic.test.ts +31 -2
- package/src/auth-email-password/web/__tests__/invite-accept-screen.test.tsx +120 -0
- package/src/auth-email-password/web/__tests__/signup-complete-screen.test.tsx +72 -0
- package/src/auth-email-password/web/auth-form-logic.ts +5 -5
- package/src/auth-email-password/web/invite-accept-screen.tsx +46 -21
- package/src/auth-email-password/web/signup-complete-screen.tsx +11 -3
- package/src/auth-mfa/__tests__/verify.integration.test.ts +6 -2
- package/src/billing-foundation/__tests__/billing-foundation.integration.test.ts +2 -0
- package/src/billing-foundation/__tests__/subscription-tier-sync.integration.test.ts +2 -0
- package/src/delivery/feature.ts +6 -1
- package/src/document-ingest-foundation/__tests__/feature.integration.test.ts +8 -1
- package/src/inbound-mail-foundation/__tests__/inbound-mail-foundation.integration.test.ts +2 -0
- package/src/inbound-mail-foundation/__tests__/retention.integration.test.ts +2 -0
- package/src/inbound-mail-foundation/__tests__/watch-supervisor.integration.test.ts +2 -0
- package/src/jobs/__tests__/tenant-job-failures.integration.test.ts +275 -0
- package/src/jobs/changes.json +7 -0
- package/src/jobs/constants.ts +1 -0
- package/src/jobs/db/queries/retention.ts +17 -1
- package/src/jobs/feature.ts +7 -1
- package/src/jobs/handlers/tenant-failures.query.ts +82 -0
- package/src/jobs/index.ts +1 -0
- package/src/jobs/job-run-logger.ts +73 -4
- package/src/jobs/tenant-job-failure-table.ts +39 -0
- package/src/rate-limiting/__tests__/rate-limiting.integration.test.ts +69 -1
- package/src/rate-limiting/changes.json +9 -1
- package/src/rate-limiting/constants.ts +1 -0
- package/src/rate-limiting/handlers/bucket-access.ts +39 -0
- package/src/rate-limiting/handlers/status.query.ts +8 -2
- package/src/subscription-mollie/__tests__/mollie-foundation.integration.test.ts +3 -0
- package/src/subscription-stripe/__tests__/stripe-foundation.integration.test.ts +4 -0
- package/src/tenant-lifecycle/__tests__/tenant-lifecycle.integration.test.ts +108 -0
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
// Tenant-visible job failures (fw#3079) end to end: a write triggers a job,
|
|
2
|
+
// the job fails, and the triggering tenant reads the failure back through
|
|
3
|
+
// `jobs:query:failures` — over real HTTP, a real BullMQ worker and a real
|
|
4
|
+
// Postgres, with the real createJobRunLogger callbacks wired in.
|
|
5
|
+
//
|
|
6
|
+
// Not setupTestStack: that helper builds a JobRunner but wires none of the
|
|
7
|
+
// bundled run-logger callbacks (test-stack.ts), which are the write path
|
|
8
|
+
// under test here. Same buildServer + createJobRunner + createJobRunLogger
|
|
9
|
+
// harness the neighbouring jobs integration tests use.
|
|
10
|
+
|
|
11
|
+
import { afterAll, beforeAll, describe, expect, test } from "bun:test";
|
|
12
|
+
import { buildServer, type JwtHelper } from "@cosmicdrift/kumiko-framework/api";
|
|
13
|
+
import { insertOne, selectMany } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
14
|
+
import type { DbConnection } from "@cosmicdrift/kumiko-framework/db";
|
|
15
|
+
import {
|
|
16
|
+
createRegistry,
|
|
17
|
+
defineFeature,
|
|
18
|
+
defineWriteHandler,
|
|
19
|
+
type SessionUser,
|
|
20
|
+
} from "@cosmicdrift/kumiko-framework/engine";
|
|
21
|
+
import { UnprocessableError } from "@cosmicdrift/kumiko-framework/errors";
|
|
22
|
+
import { createEventsTable } from "@cosmicdrift/kumiko-framework/event-store";
|
|
23
|
+
import { createJobRunner, type JobRunner } from "@cosmicdrift/kumiko-framework/jobs";
|
|
24
|
+
import {
|
|
25
|
+
createTestDb,
|
|
26
|
+
createTestRedis,
|
|
27
|
+
createTestUser,
|
|
28
|
+
type TestDb,
|
|
29
|
+
type TestRedis,
|
|
30
|
+
testTenantId,
|
|
31
|
+
unsafePushTables,
|
|
32
|
+
} from "@cosmicdrift/kumiko-framework/stack";
|
|
33
|
+
import { sleep } from "@cosmicdrift/kumiko-framework/testing";
|
|
34
|
+
import type { Hono } from "hono";
|
|
35
|
+
import { z } from "zod";
|
|
36
|
+
import { JobQueries } from "../constants";
|
|
37
|
+
import { createJobsFeature } from "../feature";
|
|
38
|
+
import { createJobRunLogger } from "../job-run-logger";
|
|
39
|
+
import { jobRunLogsTable, jobRunsTable } from "../job-run-table";
|
|
40
|
+
import { tenantJobFailuresTable } from "../tenant-job-failure-table";
|
|
41
|
+
|
|
42
|
+
const JWT_SECRET = "tenant-job-failures-integration-secret-key-0123456789";
|
|
43
|
+
const DECLARED_KEY = "app:errors.generationFailed";
|
|
44
|
+
const BUDGET_KEY = "app:errors.budgetExceeded";
|
|
45
|
+
// A provider message that must never reach the tenant.
|
|
46
|
+
const PROVIDER_MESSAGE = "openai 429: prompt 'Herr Schmidt, Kennzeichen B-XY-123' rejected";
|
|
47
|
+
|
|
48
|
+
const tenantA = testTenantId(1);
|
|
49
|
+
const tenantB = testTenantId(2);
|
|
50
|
+
const userA = createTestUser({ id: 1, tenantId: tenantA, roles: ["Admin"] });
|
|
51
|
+
const userB = createTestUser({ id: 2, tenantId: tenantB, roles: ["Admin"] });
|
|
52
|
+
const systemAdmin = createTestUser({ id: 3, tenantId: tenantA, roles: ["SystemAdmin"] });
|
|
53
|
+
|
|
54
|
+
const generateWrite = defineWriteHandler({
|
|
55
|
+
name: "generate",
|
|
56
|
+
description: "Test-only: starts the text generation for one campaign.",
|
|
57
|
+
schema: z.object({ campaignId: z.string(), mode: z.enum(["fail", "budget", "succeed"]) }),
|
|
58
|
+
access: { roles: ["Admin"] },
|
|
59
|
+
handler: async (event) => ({ isSuccess: true as const, data: { ...event.payload } }),
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
const retryWrite = defineWriteHandler({
|
|
63
|
+
name: "startFlaky",
|
|
64
|
+
description: "Test-only: starts a job that fails on every attempt.",
|
|
65
|
+
schema: z.object({}),
|
|
66
|
+
access: { roles: ["Admin"] },
|
|
67
|
+
handler: async () => ({ isSuccess: true as const, data: {} }),
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
const appFeature = defineFeature("app", (r) => {
|
|
71
|
+
r.writeHandler(generateWrite);
|
|
72
|
+
r.writeHandler(retryWrite);
|
|
73
|
+
|
|
74
|
+
r.job(
|
|
75
|
+
"generateTexts",
|
|
76
|
+
{
|
|
77
|
+
trigger: { on: "app:write:generate" },
|
|
78
|
+
tenantVisibleFailure: { messageKey: DECLARED_KEY, subjectFields: ["campaignId"] },
|
|
79
|
+
},
|
|
80
|
+
async (payload) => {
|
|
81
|
+
if (payload["mode"] === "budget") {
|
|
82
|
+
throw new UnprocessableError("budget_exceeded", { i18nKey: BUDGET_KEY });
|
|
83
|
+
}
|
|
84
|
+
if (payload["mode"] === "fail") throw new Error(PROVIDER_MESSAGE);
|
|
85
|
+
},
|
|
86
|
+
);
|
|
87
|
+
|
|
88
|
+
// retries: 1 — two attempts, both failing. Only the last one may record.
|
|
89
|
+
r.job(
|
|
90
|
+
"flaky",
|
|
91
|
+
{
|
|
92
|
+
trigger: { on: "app:write:start-flaky" },
|
|
93
|
+
retries: 1,
|
|
94
|
+
tenantVisibleFailure: { messageKey: DECLARED_KEY },
|
|
95
|
+
},
|
|
96
|
+
async () => {
|
|
97
|
+
throw new Error(PROVIDER_MESSAGE);
|
|
98
|
+
},
|
|
99
|
+
);
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
let testDb: TestDb;
|
|
103
|
+
let testRedis: TestRedis;
|
|
104
|
+
let db: DbConnection;
|
|
105
|
+
let app: Hono;
|
|
106
|
+
let jwt: JwtHelper;
|
|
107
|
+
let jobRunner: JobRunner;
|
|
108
|
+
|
|
109
|
+
beforeAll(async () => {
|
|
110
|
+
testDb = await createTestDb();
|
|
111
|
+
testRedis = await createTestRedis();
|
|
112
|
+
db = testDb.db;
|
|
113
|
+
|
|
114
|
+
const registry = createRegistry([appFeature, createJobsFeature()]);
|
|
115
|
+
await unsafePushTables(db, { jobRunsTable, jobRunLogsTable, tenantJobFailuresTable });
|
|
116
|
+
await createEventsTable(db);
|
|
117
|
+
|
|
118
|
+
const redisUrl = `redis://${testRedis.redis.options.host}:${testRedis.redis.options.port}/${testRedis.redis.options.db}`;
|
|
119
|
+
jobRunner = createJobRunner({
|
|
120
|
+
registry,
|
|
121
|
+
context: { db },
|
|
122
|
+
redisUrl,
|
|
123
|
+
consumerLane: "worker",
|
|
124
|
+
queueNamePrefix: `kumiko-tenant-job-failures-${Date.now()}`,
|
|
125
|
+
...createJobRunLogger({ db, registry }),
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
const server = buildServer({
|
|
129
|
+
registry,
|
|
130
|
+
context: { db, registry, jobRunner },
|
|
131
|
+
jwtSecret: JWT_SECRET,
|
|
132
|
+
// Event-triggered jobs enqueue from dispatch-write's afterCommit hooks,
|
|
133
|
+
// which read the runner off the dispatcher — not off the AppContext.
|
|
134
|
+
dispatcherOptions: { jobRunner },
|
|
135
|
+
});
|
|
136
|
+
app = server.app;
|
|
137
|
+
jwt = server.jwt;
|
|
138
|
+
|
|
139
|
+
await jobRunner.start();
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
afterAll(async () => {
|
|
143
|
+
await jobRunner.stop();
|
|
144
|
+
await testDb.cleanup();
|
|
145
|
+
await testRedis.cleanup();
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
async function post(path: string, user: SessionUser, body: unknown): Promise<Response> {
|
|
149
|
+
const token = await jwt.sign(user);
|
|
150
|
+
return app.request(path, {
|
|
151
|
+
method: "POST",
|
|
152
|
+
headers: { "Content-Type": "application/json", Authorization: `Bearer ${token}` },
|
|
153
|
+
body: JSON.stringify(body),
|
|
154
|
+
});
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
type FailureRow = {
|
|
158
|
+
readonly jobName: string;
|
|
159
|
+
readonly subject: Record<string, unknown> | null;
|
|
160
|
+
readonly messageKey: string;
|
|
161
|
+
};
|
|
162
|
+
|
|
163
|
+
async function generate(
|
|
164
|
+
user: SessionUser,
|
|
165
|
+
campaignId: string,
|
|
166
|
+
mode: "fail" | "budget" | "succeed",
|
|
167
|
+
): Promise<void> {
|
|
168
|
+
const res = await post("/api/write", user, {
|
|
169
|
+
type: "app:write:generate",
|
|
170
|
+
payload: { campaignId, mode },
|
|
171
|
+
});
|
|
172
|
+
expect((await res.json()).isSuccess).toBe(true);
|
|
173
|
+
await sleep(1500);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
async function failures(user: SessionUser, payload: unknown = {}): Promise<FailureRow[]> {
|
|
177
|
+
const res = await post("/api/query", user, { type: JobQueries.failures, payload });
|
|
178
|
+
const body = await res.json();
|
|
179
|
+
expect(res.status, JSON.stringify(body)).toBe(200);
|
|
180
|
+
return body.data.rows;
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
describe("jobs:query:failures (fw#3079)", () => {
|
|
184
|
+
test("the triggering tenant reads its own failed job, scoped by subject", async () => {
|
|
185
|
+
await generate(userA, "campaign-1", "fail");
|
|
186
|
+
await generate(userA, "campaign-2", "budget");
|
|
187
|
+
|
|
188
|
+
const rows = await failures(userA);
|
|
189
|
+
expect(rows).toHaveLength(2);
|
|
190
|
+
const byCampaign = new Map(rows.map((row) => [row.subject?.["campaignId"], row]));
|
|
191
|
+
expect(byCampaign.get("campaign-1")?.jobName).toBe("app:job:generate-texts");
|
|
192
|
+
// Plain Error → the key declared at the job.
|
|
193
|
+
expect(byCampaign.get("campaign-1")?.messageKey).toBe(DECLARED_KEY);
|
|
194
|
+
// KumikoError → its own i18nKey wins over the declared fallback.
|
|
195
|
+
expect(byCampaign.get("campaign-2")?.messageKey).toBe(BUDGET_KEY);
|
|
196
|
+
});
|
|
197
|
+
|
|
198
|
+
test("another tenant sees none of them, and only its own", async () => {
|
|
199
|
+
expect(await failures(userB)).toHaveLength(0);
|
|
200
|
+
|
|
201
|
+
await generate(userB, "campaign-1", "fail");
|
|
202
|
+
|
|
203
|
+
const rowsB = await failures(userB);
|
|
204
|
+
expect(rowsB).toHaveLength(1);
|
|
205
|
+
expect(rowsB[0]?.subject?.["campaignId"]).toBe("campaign-1");
|
|
206
|
+
// Tenant A's two records are untouched by B's own run of the same job
|
|
207
|
+
// and the same campaign id.
|
|
208
|
+
expect(await failures(userA)).toHaveLength(2);
|
|
209
|
+
});
|
|
210
|
+
|
|
211
|
+
test("no provider message reaches the tenant", async () => {
|
|
212
|
+
const res = await post("/api/query", userA, { type: JobQueries.failures, payload: {} });
|
|
213
|
+
const raw = await res.text();
|
|
214
|
+
expect(raw).toContain(DECLARED_KEY);
|
|
215
|
+
expect(raw).not.toContain(PROVIDER_MESSAGE);
|
|
216
|
+
expect(raw).not.toContain("openai");
|
|
217
|
+
});
|
|
218
|
+
|
|
219
|
+
test("a later successful run of the same subject clears the record", async () => {
|
|
220
|
+
await generate(userA, "campaign-1", "succeed");
|
|
221
|
+
|
|
222
|
+
const rows = await failures(userA);
|
|
223
|
+
expect(rows.map((row) => row.subject?.["campaignId"])).toEqual(["campaign-2"]);
|
|
224
|
+
});
|
|
225
|
+
|
|
226
|
+
test("only the final attempt records — a retried job leaves one record", async () => {
|
|
227
|
+
const res = await post("/api/write", userB, { type: "app:write:start-flaky", payload: {} });
|
|
228
|
+
expect((await res.json()).isSuccess).toBe(true);
|
|
229
|
+
await sleep(2500);
|
|
230
|
+
|
|
231
|
+
const rows = await failures(userB, { jobName: "app:job:flaky" });
|
|
232
|
+
expect(rows).toHaveLength(1);
|
|
233
|
+
expect(rows[0]?.subject).toBeNull();
|
|
234
|
+
|
|
235
|
+
// Both attempts did land as their own failed run — the single record
|
|
236
|
+
// above is the final-attempt gate, not a missing second attempt.
|
|
237
|
+
const runs = await selectMany(db, jobRunsTable, { jobName: "app:job:flaky" });
|
|
238
|
+
expect(runs).toHaveLength(2);
|
|
239
|
+
});
|
|
240
|
+
|
|
241
|
+
test("the subject filter selects one record", async () => {
|
|
242
|
+
const rows = await failures(userA, { subject: { campaignId: "campaign-2" } });
|
|
243
|
+
expect(rows).toHaveLength(1);
|
|
244
|
+
expect(rows[0]?.messageKey).toBe(BUDGET_KEY);
|
|
245
|
+
|
|
246
|
+
expect(await failures(userA, { subject: { campaignId: "campaign-unknown" } })).toHaveLength(0);
|
|
247
|
+
});
|
|
248
|
+
|
|
249
|
+
test("SystemAdmin still sees every tenant's run with its provider message", async () => {
|
|
250
|
+
const res = await post("/api/query", systemAdmin, {
|
|
251
|
+
type: JobQueries.list,
|
|
252
|
+
payload: { jobName: "app:job:generate-texts", status: "failed" },
|
|
253
|
+
});
|
|
254
|
+
const raw = await res.text();
|
|
255
|
+
expect(res.status, raw).toBe(200);
|
|
256
|
+
|
|
257
|
+
// Both of tenant A's failures plus tenant B's — the tenant-visible record
|
|
258
|
+
// is an addition, it takes nothing away from the SystemAdmin view.
|
|
259
|
+
expect(JSON.parse(raw).data.rows).toHaveLength(3);
|
|
260
|
+
expect(raw).toContain(PROVIDER_MESSAGE);
|
|
261
|
+
});
|
|
262
|
+
test("a corrupt stored subject degrades to null instead of failing the list", async () => {
|
|
263
|
+
await insertOne(db, tenantJobFailuresTable, {
|
|
264
|
+
tenantId: tenantB,
|
|
265
|
+
jobName: "app:job:corrupt",
|
|
266
|
+
subject: '{"campaignId":"not-an-entry-array"}',
|
|
267
|
+
messageKey: DECLARED_KEY,
|
|
268
|
+
failedAt: Temporal.Now.instant(),
|
|
269
|
+
});
|
|
270
|
+
|
|
271
|
+
const rows = await failures(userB, { jobName: "app:job:corrupt" });
|
|
272
|
+
expect(rows).toHaveLength(1);
|
|
273
|
+
expect(rows[0]?.subject).toBeNull();
|
|
274
|
+
});
|
|
275
|
+
});
|
package/src/jobs/changes.json
CHANGED
|
@@ -1,4 +1,11 @@
|
|
|
1
1
|
[
|
|
2
|
+
{
|
|
3
|
+
"version": "0.291.0",
|
|
4
|
+
"type": "improvement",
|
|
5
|
+
"title": "Tenant-visible job failures: `r.job({ tenantVisibleFailure })` plus `jobs:query:failures` (fw#3079)",
|
|
6
|
+
"detail": "A fire-and-forget job that fails left the tenant's screen on a spinner that never ends — `jobs:query:list` is SystemAdmin and reads cross-tenant over `systemDb.unsafeRaw`, so a tenant could not see its own job failing. Apps worked around it with their own failure entity written in the job's catch.\nA job now opts in declaratively: `r.job(\"generateTexts\", { trigger: …, tenantVisibleFailure: { messageKey: \"app:errors.generationFailed\", subjectFields: [\"campaignId\"] } }, handler)`. When its last attempt fails, the run-logger records one row per tenant, job and subject in the new `store_tenant_job_failures` table, and the tenant reads it back through `jobs:query:failures` (every membership rank, own tenant only).\nOnly a translation key travels to the tenant: the thrown error's own `i18nKey` when it carries one, otherwise the declared `messageKey`. The provider's message stays on `store_job_runs.error` and in the run log, both SystemAdmin-only. Records are scoped to the run's tenant — a tenant-less run (cron, `SYSTEM_TENANT_ID`) records nothing.\nLifetime and retries: a record lives until the next successful run of the same job and subject deletes it; there is no acknowledgement step (tenant job administration stays out of scope). Only the final attempt records, so a job with `retries` that succeeds on a later attempt never shows the tenant a failure. The daily `retention-cleanup` job purges leftovers past `retentionDays`.\n`jobs:query:list`, `jobs:query:details` and `jobs:query:retry` are unchanged. `JobRunnerOptions.onJobComplete`/`onJobFailed` gained an optional fifth `outcome` argument — existing four-argument callbacks keep working.",
|
|
7
|
+
"migration": "New store table. Run `kumiko migrate generate` and apply the migration — `store_tenant_job_failures` is created empty and stays empty until a job declares `tenantVisibleFailure`. No change needed for apps that do not opt in."
|
|
8
|
+
},
|
|
2
9
|
{
|
|
3
10
|
"version": "0.241.0",
|
|
4
11
|
"type": "breaking",
|
package/src/jobs/constants.ts
CHANGED
|
@@ -13,12 +13,14 @@
|
|
|
13
13
|
import { deleteManyBatched } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
14
14
|
import type { DbConnection } from "@cosmicdrift/kumiko-framework/db";
|
|
15
15
|
import { jobRunLogsTable, jobRunsTable } from "../../job-run-table";
|
|
16
|
+
import { tenantJobFailuresTable } from "../../tenant-job-failure-table";
|
|
16
17
|
|
|
17
18
|
const RETENTION_DELETE_BATCH_SIZE = 500;
|
|
18
19
|
|
|
19
20
|
export type JobRunRetentionResult = {
|
|
20
21
|
readonly runsDeleted: number;
|
|
21
22
|
readonly logsDeleted: number;
|
|
23
|
+
readonly tenantFailuresDeleted: number;
|
|
22
24
|
};
|
|
23
25
|
|
|
24
26
|
export async function deleteStaleJobRuns(
|
|
@@ -40,5 +42,19 @@ export async function deleteStaleJobRuns(
|
|
|
40
42
|
{ limit: RETENTION_DELETE_BATCH_SIZE },
|
|
41
43
|
);
|
|
42
44
|
|
|
43
|
-
|
|
45
|
+
// Same window for the tenant-visible failure records (fw#3079): the run
|
|
46
|
+
// they point at is gone by now, and a job whose subject never ran again
|
|
47
|
+
// would otherwise keep its record forever.
|
|
48
|
+
const tenantFailuresResult = await deleteManyBatched(
|
|
49
|
+
db,
|
|
50
|
+
tenantJobFailuresTable,
|
|
51
|
+
{ failedAt: { lt: cutoff } },
|
|
52
|
+
{ limit: RETENTION_DELETE_BATCH_SIZE },
|
|
53
|
+
);
|
|
54
|
+
|
|
55
|
+
return {
|
|
56
|
+
runsDeleted: runsResult.deleted,
|
|
57
|
+
logsDeleted: logsResult.deleted,
|
|
58
|
+
tenantFailuresDeleted: tenantFailuresResult.deleted,
|
|
59
|
+
};
|
|
44
60
|
}
|
package/src/jobs/feature.ts
CHANGED
|
@@ -21,9 +21,11 @@ import {
|
|
|
21
21
|
createStaleRunSweepJob,
|
|
22
22
|
DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS,
|
|
23
23
|
} from "./handlers/stale-run-sweep.job";
|
|
24
|
+
import { tenantFailuresQuery } from "./handlers/tenant-failures.query";
|
|
24
25
|
import { triggerWrite } from "./handlers/trigger.write";
|
|
25
26
|
import { JOBS_I18N } from "./i18n";
|
|
26
27
|
import { jobRunLogsTableMeta, jobRunsTableMeta } from "./job-run-table";
|
|
28
|
+
import { tenantJobFailuresTableMeta } from "./tenant-job-failure-table";
|
|
27
29
|
|
|
28
30
|
export type JobsFeatureOptions = {
|
|
29
31
|
// How long a job run (and its logs) stays in store_job_runs/
|
|
@@ -42,7 +44,7 @@ export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefi
|
|
|
42
44
|
const staleRunTimeoutHours = options.staleRunTimeoutHours ?? DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS;
|
|
43
45
|
return defineFeature("jobs", (r) => {
|
|
44
46
|
r.describe(
|
|
45
|
-
"Persistence and operator tooling for background jobs registered via `r.job(...)`. Every job execution writes directly into `store_job_runs` (current status + duration) and `store_job_run_logs` (per-line log rows) from the BullMQ callbacks — no event stream in between (#2243). A daily `retention-cleanup` job deletes runs (and their logs) older than `retentionDays`; an hourly `stale-run-sweep` job marks runs stuck at status `running` past `staleRunTimeoutHours` as `failed` (#2246 — a crashed worker never fires the completion callback, so nothing else ever revisits the row). Exposes `jobs:write:trigger` (manual run) and `jobs:write:retry` (operator retry of a failed run), plus `jobs:query:list`, `jobs:query:details`, and `jobs:query:catalog` (manual jobs) for the operator UI.",
|
|
47
|
+
"Persistence and operator tooling for background jobs registered via `r.job(...)`. Every job execution writes directly into `store_job_runs` (current status + duration) and `store_job_run_logs` (per-line log rows) from the BullMQ callbacks — no event stream in between (#2243). A daily `retention-cleanup` job deletes runs (and their logs) older than `retentionDays`; an hourly `stale-run-sweep` job marks runs stuck at status `running` past `staleRunTimeoutHours` as `failed` (#2246 — a crashed worker never fires the completion callback, so nothing else ever revisits the row). Exposes `jobs:write:trigger` (manual run) and `jobs:write:retry` (operator retry of a failed run), plus `jobs:query:list`, `jobs:query:details`, and `jobs:query:catalog` (manual jobs) for the operator UI. A job that declares `tenantVisibleFailure` also records its last failed attempt per tenant and subject in `store_tenant_job_failures`, which the tenant itself reads through `jobs:query:failures` — a translation key only, never the provider's message (fw#3079).",
|
|
46
48
|
);
|
|
47
49
|
r.uiHints({
|
|
48
50
|
displayLabel: "Jobs · Audit & Operator UI",
|
|
@@ -61,6 +63,9 @@ export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefi
|
|
|
61
63
|
r.storeTable(jobRunLogsTableMeta, {
|
|
62
64
|
reason: "read_side.job_run_logs",
|
|
63
65
|
});
|
|
66
|
+
r.storeTable(tenantJobFailuresTableMeta, {
|
|
67
|
+
reason: "direct_write.tenant_job_failures",
|
|
68
|
+
});
|
|
64
69
|
|
|
65
70
|
// Framework-provided rebuild job — available whenever `jobs` is composed; enqueueProjectionRebuild dispatches it.
|
|
66
71
|
r.job(
|
|
@@ -112,6 +117,7 @@ export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefi
|
|
|
112
117
|
list: r.queryHandler(listQuery),
|
|
113
118
|
detail: r.queryHandler(detailQuery),
|
|
114
119
|
catalog: r.queryHandler(catalogQuery),
|
|
120
|
+
failures: r.queryHandler(tenantFailuresQuery),
|
|
115
121
|
};
|
|
116
122
|
|
|
117
123
|
const systemAdminAccess = { roles: ["SystemAdmin"] as const };
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
import { selectMany } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
2
|
+
import { access, defineQueryHandler } from "@cosmicdrift/kumiko-framework/engine";
|
|
3
|
+
import { InternalError } from "@cosmicdrift/kumiko-framework/errors";
|
|
4
|
+
import { parseJsonSafe } from "@cosmicdrift/kumiko-framework/utils";
|
|
5
|
+
import { z } from "zod";
|
|
6
|
+
import { tenantJobFailuresTable } from "../tenant-job-failure-table";
|
|
7
|
+
|
|
8
|
+
type TenantJobFailureRow = {
|
|
9
|
+
readonly tenantId: string;
|
|
10
|
+
readonly jobName: string;
|
|
11
|
+
readonly subject: string | null;
|
|
12
|
+
readonly messageKey: string;
|
|
13
|
+
readonly failedAt: Temporal.Instant;
|
|
14
|
+
};
|
|
15
|
+
|
|
16
|
+
const DEFAULT_LIMIT = 50;
|
|
17
|
+
|
|
18
|
+
// Mirrors job-runner.ts's jobSubjectKey: sorted field names, so a caller that
|
|
19
|
+
// passes the same subject values gets the same string the writer stored.
|
|
20
|
+
function subjectKey(subject: Record<string, string | number | boolean | null>): string {
|
|
21
|
+
return JSON.stringify(
|
|
22
|
+
Object.keys(subject)
|
|
23
|
+
.sort()
|
|
24
|
+
.map((field) => [field, subject[field] ?? null]),
|
|
25
|
+
);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
function isSubjectEntry(value: unknown): value is [string, unknown] {
|
|
29
|
+
return Array.isArray(value) && typeof value[0] === "string";
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function parseSubject(stored: string | null): Record<string, unknown> | null {
|
|
33
|
+
if (stored === null) return null;
|
|
34
|
+
// A corrupt subject must not fail the whole list — the record still tells
|
|
35
|
+
// the tenant which job failed and why. parseJsonSafe only survives a
|
|
36
|
+
// SyntaxError, so the shape needs its own check.
|
|
37
|
+
const parsed = parseJsonSafe<unknown>(stored, null);
|
|
38
|
+
if (!Array.isArray(parsed)) return null;
|
|
39
|
+
return Object.fromEntries(parsed.filter(isSubjectEntry));
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export const tenantFailuresQuery = defineQueryHandler({
|
|
43
|
+
name: "failures",
|
|
44
|
+
description:
|
|
45
|
+
"Lists the calling tenant's own failed jobs — one record per job and subject, newest first, each carrying a translation key for the reason, never the provider's own error message; use it to tell a tenant that their asynchronous job failed instead of leaving the screen waiting.",
|
|
46
|
+
schema: z.object({
|
|
47
|
+
jobName: z.string().optional(),
|
|
48
|
+
subject: z.record(z.string(), z.union([z.string(), z.number(), z.boolean()])).optional(),
|
|
49
|
+
limit: z.number().min(1).max(200).optional(),
|
|
50
|
+
}),
|
|
51
|
+
// Every membership rank: a failure record carries a job name and a
|
|
52
|
+
// translation key, nothing a team member of the tenant may not see.
|
|
53
|
+
access: { roles: access.roles("User", "Editor", ...access.admin) },
|
|
54
|
+
handler: async (query, ctx) => {
|
|
55
|
+
if (!ctx.systemDb) {
|
|
56
|
+
throw new InternalError({
|
|
57
|
+
message: "jobs:query:failures requires ctx.systemDb (feature must declare r.systemScope())",
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
// assertTenantMatch is a self-check on the caller and returns the same
|
|
61
|
+
// unfiltered system-mode db (tenant-db.ts) — the explicit tenantId below
|
|
62
|
+
// is the filter, assertRowsTenant the second net. There is no
|
|
63
|
+
// cross-tenant mode here: SystemAdmin reads runs via jobs:query:list.
|
|
64
|
+
const db = ctx.systemDb.assertTenantMatch(query.user.tenantId);
|
|
65
|
+
const where: Record<string, unknown> = { tenantId: [query.user.tenantId] };
|
|
66
|
+
if (query.payload.jobName) where["jobName"] = query.payload.jobName;
|
|
67
|
+
if (query.payload.subject) where["subject"] = subjectKey(query.payload.subject);
|
|
68
|
+
const rows = await selectMany<TenantJobFailureRow>(db, tenantJobFailuresTable, where, {
|
|
69
|
+
orderBy: { col: "failedAt", direction: "desc" },
|
|
70
|
+
limit: query.payload.limit ?? DEFAULT_LIMIT,
|
|
71
|
+
});
|
|
72
|
+
return {
|
|
73
|
+
rows: ctx.systemDb.assertRowsTenant(rows, "tenantId").map((row) => ({
|
|
74
|
+
jobName: row.jobName,
|
|
75
|
+
subject: parseSubject(row.subject),
|
|
76
|
+
messageKey: row.messageKey,
|
|
77
|
+
failedAt: row.failedAt,
|
|
78
|
+
})),
|
|
79
|
+
nextCursor: null,
|
|
80
|
+
};
|
|
81
|
+
},
|
|
82
|
+
});
|
package/src/jobs/index.ts
CHANGED
|
@@ -4,3 +4,4 @@ export type { JobRunLoggerCallbacks } from "./job-run-logger";
|
|
|
4
4
|
export { createJobRunLogger } from "./job-run-logger";
|
|
5
5
|
export type { JobLogLevel, JobRunStatus } from "./job-run-table";
|
|
6
6
|
export { jobRunLogsTable, jobRunsTable } from "./job-run-table";
|
|
7
|
+
export { tenantJobFailuresTable } from "./tenant-job-failure-table";
|
|
@@ -1,4 +1,10 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import {
|
|
2
|
+
deleteMany,
|
|
3
|
+
fetchOne,
|
|
4
|
+
insertMany,
|
|
5
|
+
insertOne,
|
|
6
|
+
updateMany,
|
|
7
|
+
} from "@cosmicdrift/kumiko-framework/bun-db";
|
|
2
8
|
import {
|
|
3
9
|
configuredPiiSubjectKms,
|
|
4
10
|
encryptPiiValueForSubject,
|
|
@@ -8,12 +14,18 @@ import {
|
|
|
8
14
|
} from "@cosmicdrift/kumiko-framework/crypto";
|
|
9
15
|
import type { DbConnection } from "@cosmicdrift/kumiko-framework/db";
|
|
10
16
|
import { type Registry, SYSTEM_TENANT_ID } from "@cosmicdrift/kumiko-framework/engine";
|
|
11
|
-
import type {
|
|
17
|
+
import type {
|
|
18
|
+
JobLogEntry,
|
|
19
|
+
JobMeta,
|
|
20
|
+
JobOutcomeMeta,
|
|
21
|
+
JobRunnerOptions,
|
|
22
|
+
} from "@cosmicdrift/kumiko-framework/jobs";
|
|
12
23
|
import { generateId } from "@cosmicdrift/kumiko-framework/utils";
|
|
13
24
|
import { mapWithConcurrency } from "../shared";
|
|
14
25
|
import { runCompletedSchema, runFailedSchema, runStartedSchema } from "./events";
|
|
15
26
|
import { parseJobInstant } from "./job-instant";
|
|
16
27
|
import { jobRunLogsTable, jobRunsTable } from "./job-run-table";
|
|
28
|
+
import { tenantJobFailuresTable } from "./tenant-job-failure-table";
|
|
17
29
|
|
|
18
30
|
// Matches PgKmsAdapter's default pool size (see tenant/handlers/*.query.ts) —
|
|
19
31
|
// bounds concurrent getOrCreateDek calls so a large log batch doesn't claim
|
|
@@ -138,6 +150,54 @@ async function encryptStartedPayload(
|
|
|
138
150
|
return encryptOrSentinel(kms, triggeredById, payload, "payload");
|
|
139
151
|
}
|
|
140
152
|
|
|
153
|
+
// fw#3079 — which row the tenant-visible failure record lives in: one per
|
|
154
|
+
// (tenant, job, subject), so the next outcome of the same work replaces or
|
|
155
|
+
// clears it. `subject` is null for a job that declares no subjectFields.
|
|
156
|
+
// Null target = nothing to write or clear: the job did not opt in, the run
|
|
157
|
+
// was tenant-less (cron resolves to SYSTEM_TENANT_ID, where no tenant-scoped
|
|
158
|
+
// query could ever read the row), or the caller predates fw#3079 and passes
|
|
159
|
+
// no outcome at all.
|
|
160
|
+
function tenantJobFailureTarget(
|
|
161
|
+
jobName: string,
|
|
162
|
+
outcome: JobOutcomeMeta | undefined,
|
|
163
|
+
): Record<string, unknown> | null {
|
|
164
|
+
if (!outcome?.tenantVisible || outcome.tenantId === SYSTEM_TENANT_ID) return null;
|
|
165
|
+
return { tenantId: outcome.tenantId, jobName, subject: outcome.tenantVisible.subject };
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
async function recordTenantJobFailure(
|
|
169
|
+
db: DbConnection,
|
|
170
|
+
jobName: string,
|
|
171
|
+
outcome: JobOutcomeMeta | undefined,
|
|
172
|
+
): Promise<void> {
|
|
173
|
+
const where = tenantJobFailureTarget(jobName, outcome);
|
|
174
|
+
const messageKey = outcome?.tenantVisible?.messageKey;
|
|
175
|
+
// skip: no tenant-visible target, or an attempt BullMQ may still retry — a
|
|
176
|
+
// non-final failure must not show the tenant a failure the next attempt
|
|
177
|
+
// may still resolve.
|
|
178
|
+
if (!where || !messageKey || outcome?.finalAttempt !== true) return;
|
|
179
|
+
// ponytail: delete-then-insert instead of an upsert — two runs of the same
|
|
180
|
+
// key finishing at once can leave two rows, and the query returns the
|
|
181
|
+
// newest. Add a unique index + ON CONFLICT if that ever matters.
|
|
182
|
+
await deleteMany(db, tenantJobFailuresTable, where);
|
|
183
|
+
await insertOne(db, tenantJobFailuresTable, {
|
|
184
|
+
...where,
|
|
185
|
+
messageKey,
|
|
186
|
+
failedAt: Temporal.Now.instant(),
|
|
187
|
+
});
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
async function clearTenantJobFailure(
|
|
191
|
+
db: DbConnection,
|
|
192
|
+
jobName: string,
|
|
193
|
+
outcome: JobOutcomeMeta | undefined,
|
|
194
|
+
): Promise<void> {
|
|
195
|
+
const where = tenantJobFailureTarget(jobName, outcome);
|
|
196
|
+
// skip: no tenant-visible target — nothing was ever recorded
|
|
197
|
+
if (!where) return;
|
|
198
|
+
await deleteMany(db, tenantJobFailuresTable, where);
|
|
199
|
+
}
|
|
200
|
+
|
|
141
201
|
export function createJobRunLogger(opts: JobRunLoggerOptions): JobRunLoggerCallbacks {
|
|
142
202
|
const { db } = opts;
|
|
143
203
|
|
|
@@ -241,11 +301,16 @@ export function createJobRunLogger(opts: JobRunLoggerOptions): JobRunLoggerCallb
|
|
|
241
301
|
},
|
|
242
302
|
|
|
243
303
|
onJobComplete: async (
|
|
244
|
-
|
|
304
|
+
jobName: string,
|
|
245
305
|
bullJobId: string,
|
|
246
306
|
duration: number,
|
|
247
307
|
logs: JobLogEntry[],
|
|
308
|
+
outcome?: JobOutcomeMeta,
|
|
248
309
|
) => {
|
|
310
|
+
// Before the run-row write and independent of it: a successful run
|
|
311
|
+
// clears the tenant's failure record even when the run row itself is
|
|
312
|
+
// unreachable (the state-loss return below).
|
|
313
|
+
await clearTenantJobFailure(db, jobName, outcome);
|
|
249
314
|
const resolved = await resolveRun(bullJobId);
|
|
250
315
|
// skip: state loss between start + complete (worker restart, cache
|
|
251
316
|
// evicted AND DB has no matching bull_job_id). Rare edge case; we
|
|
@@ -297,11 +362,15 @@ export function createJobRunLogger(opts: JobRunLoggerOptions): JobRunLoggerCallb
|
|
|
297
362
|
},
|
|
298
363
|
|
|
299
364
|
onJobFailed: async (
|
|
300
|
-
|
|
365
|
+
jobName: string,
|
|
301
366
|
bullJobId: string,
|
|
302
367
|
error: string,
|
|
303
368
|
logs: JobLogEntry[],
|
|
369
|
+
outcome?: JobOutcomeMeta,
|
|
304
370
|
) => {
|
|
371
|
+
// Mirror of onJobComplete: recorded independently of the run row, so a
|
|
372
|
+
// tenant still learns their job failed if the row is unreachable.
|
|
373
|
+
await recordTenantJobFailure(db, jobName, outcome);
|
|
305
374
|
const resolved = await resolveRun(bullJobId);
|
|
306
375
|
// skip: same rare state-loss case as in onJobComplete — drop the
|
|
307
376
|
// failure write rather than forge a run row from scratch.
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { asEntityTableMeta } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
2
|
+
import {
|
|
3
|
+
type EntityTableMeta,
|
|
4
|
+
instant,
|
|
5
|
+
table as pgTable,
|
|
6
|
+
serial,
|
|
7
|
+
sql,
|
|
8
|
+
text,
|
|
9
|
+
uuid,
|
|
10
|
+
} from "@cosmicdrift/kumiko-framework/db";
|
|
11
|
+
|
|
12
|
+
// The tenant's own view of a failed job (fw#3079). Direct-write store like
|
|
13
|
+
// store_job_runs: job-run-logger.ts writes it from the BullMQ callbacks,
|
|
14
|
+
// outside any dispatcher transaction.
|
|
15
|
+
//
|
|
16
|
+
// Deliberately holds no error text. `message_key` is a translation key
|
|
17
|
+
// (the thrown KumikoError's i18nKey or the one declared at the job), so a
|
|
18
|
+
// provider message can never reach the tenant through this table — it stays
|
|
19
|
+
// on store_job_runs.error and in store_job_run_logs, both SystemAdmin-only.
|
|
20
|
+
//
|
|
21
|
+
// One row per (tenant, job, subject): a later final-attempt failure replaces
|
|
22
|
+
// it, a later successful run of the same key deletes it. `subject` is the
|
|
23
|
+
// canonical JSON of the job's declared subjectFields, or NULL for a job that
|
|
24
|
+
// declares none — stored in clear, so a job must not declare a PII field
|
|
25
|
+
// as its subject.
|
|
26
|
+
export const tenantJobFailuresTable = pgTable("store_tenant_job_failures", {
|
|
27
|
+
id: serial("id").primaryKey(),
|
|
28
|
+
tenantId: uuid("tenant_id").notNull(),
|
|
29
|
+
jobName: text("job_name").notNull(),
|
|
30
|
+
subject: text("subject"),
|
|
31
|
+
messageKey: text("message_key").notNull(),
|
|
32
|
+
failedAt: instant("failed_at").default(sql`now()`).notNull(),
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
const derivedTenantJobFailuresTableMeta = asEntityTableMeta(tenantJobFailuresTable);
|
|
36
|
+
if (!derivedTenantJobFailuresTableMeta) {
|
|
37
|
+
throw new Error("tenantJobFailuresTable: table carries no EntityTableMeta — built via table()?");
|
|
38
|
+
}
|
|
39
|
+
export const tenantJobFailuresTableMeta: EntityTableMeta = derivedTenantJobFailuresTableMeta;
|