@cosmicdrift/kumiko-bundled-features 0.290.0 → 0.292.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +9 -9
- package/src/audit/__tests__/audit-screens.boot.test.ts +40 -1
- package/src/audit/__tests__/audit.integration.test.ts +96 -2
- package/src/audit/__tests__/escape-hatch-audit.integration.test.ts +8 -1
- package/src/audit/changes.json +7 -0
- package/src/audit/feature.ts +18 -4
- package/src/auth-email-password/changes.json +6 -0
- package/src/auth-email-password/web/__tests__/auth-form-logic.test.ts +31 -2
- package/src/auth-email-password/web/__tests__/invite-accept-screen.test.tsx +120 -0
- package/src/auth-email-password/web/__tests__/signup-complete-screen.test.tsx +72 -0
- package/src/auth-email-password/web/auth-form-logic.ts +5 -5
- package/src/auth-email-password/web/invite-accept-screen.tsx +46 -21
- package/src/auth-email-password/web/signup-complete-screen.tsx +11 -3
- package/src/custom-fields/__tests__/audit-integration.integration.test.ts +2 -0
- package/src/delivery/feature.ts +6 -1
- package/src/file-foundation/__tests__/file-foundation.integration.test.ts +132 -9
- package/src/file-foundation/changes.json +9 -1
- package/src/file-foundation/feature.ts +2 -0
- package/src/jobs/__tests__/tenant-job-failures.integration.test.ts +275 -0
- package/src/jobs/changes.json +7 -0
- package/src/jobs/constants.ts +1 -0
- package/src/jobs/db/queries/retention.ts +17 -1
- package/src/jobs/feature.ts +7 -1
- package/src/jobs/handlers/tenant-failures.query.ts +82 -0
- package/src/jobs/index.ts +1 -0
- package/src/jobs/job-run-logger.ts +73 -4
- package/src/jobs/tenant-job-failure-table.ts +39 -0
- package/src/rate-limiting/__tests__/rate-limiting.integration.test.ts +69 -1
- package/src/rate-limiting/changes.json +9 -1
- package/src/rate-limiting/constants.ts +1 -0
- package/src/rate-limiting/handlers/bucket-access.ts +39 -0
- package/src/rate-limiting/handlers/status.query.ts +8 -2
package/src/jobs/changes.json
CHANGED
|
@@ -1,4 +1,11 @@
|
|
|
1
1
|
[
|
|
2
|
+
{
|
|
3
|
+
"version": "0.291.0",
|
|
4
|
+
"type": "improvement",
|
|
5
|
+
"title": "Tenant-visible job failures: `r.job({ tenantVisibleFailure })` plus `jobs:query:failures` (fw#3079)",
|
|
6
|
+
"detail": "A fire-and-forget job that fails left the tenant's screen on a spinner that never ends — `jobs:query:list` is SystemAdmin and reads cross-tenant over `systemDb.unsafeRaw`, so a tenant could not see its own job failing. Apps worked around it with their own failure entity written in the job's catch.\nA job now opts in declaratively: `r.job(\"generateTexts\", { trigger: …, tenantVisibleFailure: { messageKey: \"app:errors.generationFailed\", subjectFields: [\"campaignId\"] } }, handler)`. When its last attempt fails, the run-logger records one row per tenant, job and subject in the new `store_tenant_job_failures` table, and the tenant reads it back through `jobs:query:failures` (every membership rank, own tenant only).\nOnly a translation key travels to the tenant: the thrown error's own `i18nKey` when it carries one, otherwise the declared `messageKey`. The provider's message stays on `store_job_runs.error` and in the run log, both SystemAdmin-only. Records are scoped to the run's tenant — a tenant-less run (cron, `SYSTEM_TENANT_ID`) records nothing.\nLifetime and retries: a record lives until the next successful run of the same job and subject deletes it; there is no acknowledgement step (tenant job administration stays out of scope). Only the final attempt records, so a job with `retries` that succeeds on a later attempt never shows the tenant a failure. The daily `retention-cleanup` job purges leftovers past `retentionDays`.\n`jobs:query:list`, `jobs:query:details` and `jobs:query:retry` are unchanged. `JobRunnerOptions.onJobComplete`/`onJobFailed` gained an optional fifth `outcome` argument — existing four-argument callbacks keep working.",
|
|
7
|
+
"migration": "New store table. Run `kumiko migrate generate` and apply the migration — `store_tenant_job_failures` is created empty and stays empty until a job declares `tenantVisibleFailure`. No change needed for apps that do not opt in."
|
|
8
|
+
},
|
|
2
9
|
{
|
|
3
10
|
"version": "0.241.0",
|
|
4
11
|
"type": "breaking",
|
package/src/jobs/constants.ts
CHANGED
|
@@ -13,12 +13,14 @@
|
|
|
13
13
|
import { deleteManyBatched } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
14
14
|
import type { DbConnection } from "@cosmicdrift/kumiko-framework/db";
|
|
15
15
|
import { jobRunLogsTable, jobRunsTable } from "../../job-run-table";
|
|
16
|
+
import { tenantJobFailuresTable } from "../../tenant-job-failure-table";
|
|
16
17
|
|
|
17
18
|
const RETENTION_DELETE_BATCH_SIZE = 500;
|
|
18
19
|
|
|
19
20
|
export type JobRunRetentionResult = {
|
|
20
21
|
readonly runsDeleted: number;
|
|
21
22
|
readonly logsDeleted: number;
|
|
23
|
+
readonly tenantFailuresDeleted: number;
|
|
22
24
|
};
|
|
23
25
|
|
|
24
26
|
export async function deleteStaleJobRuns(
|
|
@@ -40,5 +42,19 @@ export async function deleteStaleJobRuns(
|
|
|
40
42
|
{ limit: RETENTION_DELETE_BATCH_SIZE },
|
|
41
43
|
);
|
|
42
44
|
|
|
43
|
-
|
|
45
|
+
// Same window for the tenant-visible failure records (fw#3079): the run
|
|
46
|
+
// they point at is gone by now, and a job whose subject never ran again
|
|
47
|
+
// would otherwise keep its record forever.
|
|
48
|
+
const tenantFailuresResult = await deleteManyBatched(
|
|
49
|
+
db,
|
|
50
|
+
tenantJobFailuresTable,
|
|
51
|
+
{ failedAt: { lt: cutoff } },
|
|
52
|
+
{ limit: RETENTION_DELETE_BATCH_SIZE },
|
|
53
|
+
);
|
|
54
|
+
|
|
55
|
+
return {
|
|
56
|
+
runsDeleted: runsResult.deleted,
|
|
57
|
+
logsDeleted: logsResult.deleted,
|
|
58
|
+
tenantFailuresDeleted: tenantFailuresResult.deleted,
|
|
59
|
+
};
|
|
44
60
|
}
|
package/src/jobs/feature.ts
CHANGED
|
@@ -21,9 +21,11 @@ import {
|
|
|
21
21
|
createStaleRunSweepJob,
|
|
22
22
|
DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS,
|
|
23
23
|
} from "./handlers/stale-run-sweep.job";
|
|
24
|
+
import { tenantFailuresQuery } from "./handlers/tenant-failures.query";
|
|
24
25
|
import { triggerWrite } from "./handlers/trigger.write";
|
|
25
26
|
import { JOBS_I18N } from "./i18n";
|
|
26
27
|
import { jobRunLogsTableMeta, jobRunsTableMeta } from "./job-run-table";
|
|
28
|
+
import { tenantJobFailuresTableMeta } from "./tenant-job-failure-table";
|
|
27
29
|
|
|
28
30
|
export type JobsFeatureOptions = {
|
|
29
31
|
// How long a job run (and its logs) stays in store_job_runs/
|
|
@@ -42,7 +44,7 @@ export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefi
|
|
|
42
44
|
const staleRunTimeoutHours = options.staleRunTimeoutHours ?? DEFAULT_JOB_RUN_STALE_TIMEOUT_HOURS;
|
|
43
45
|
return defineFeature("jobs", (r) => {
|
|
44
46
|
r.describe(
|
|
45
|
-
"Persistence and operator tooling for background jobs registered via `r.job(...)`. Every job execution writes directly into `store_job_runs` (current status + duration) and `store_job_run_logs` (per-line log rows) from the BullMQ callbacks — no event stream in between (#2243). A daily `retention-cleanup` job deletes runs (and their logs) older than `retentionDays`; an hourly `stale-run-sweep` job marks runs stuck at status `running` past `staleRunTimeoutHours` as `failed` (#2246 — a crashed worker never fires the completion callback, so nothing else ever revisits the row). Exposes `jobs:write:trigger` (manual run) and `jobs:write:retry` (operator retry of a failed run), plus `jobs:query:list`, `jobs:query:details`, and `jobs:query:catalog` (manual jobs) for the operator UI.",
|
|
47
|
+
"Persistence and operator tooling for background jobs registered via `r.job(...)`. Every job execution writes directly into `store_job_runs` (current status + duration) and `store_job_run_logs` (per-line log rows) from the BullMQ callbacks — no event stream in between (#2243). A daily `retention-cleanup` job deletes runs (and their logs) older than `retentionDays`; an hourly `stale-run-sweep` job marks runs stuck at status `running` past `staleRunTimeoutHours` as `failed` (#2246 — a crashed worker never fires the completion callback, so nothing else ever revisits the row). Exposes `jobs:write:trigger` (manual run) and `jobs:write:retry` (operator retry of a failed run), plus `jobs:query:list`, `jobs:query:details`, and `jobs:query:catalog` (manual jobs) for the operator UI. A job that declares `tenantVisibleFailure` also records its last failed attempt per tenant and subject in `store_tenant_job_failures`, which the tenant itself reads through `jobs:query:failures` — a translation key only, never the provider's message (fw#3079).",
|
|
46
48
|
);
|
|
47
49
|
r.uiHints({
|
|
48
50
|
displayLabel: "Jobs · Audit & Operator UI",
|
|
@@ -61,6 +63,9 @@ export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefi
|
|
|
61
63
|
r.storeTable(jobRunLogsTableMeta, {
|
|
62
64
|
reason: "read_side.job_run_logs",
|
|
63
65
|
});
|
|
66
|
+
r.storeTable(tenantJobFailuresTableMeta, {
|
|
67
|
+
reason: "direct_write.tenant_job_failures",
|
|
68
|
+
});
|
|
64
69
|
|
|
65
70
|
// Framework-provided rebuild job — available whenever `jobs` is composed; enqueueProjectionRebuild dispatches it.
|
|
66
71
|
r.job(
|
|
@@ -112,6 +117,7 @@ export function createJobsFeature(options: JobsFeatureOptions = {}): FeatureDefi
|
|
|
112
117
|
list: r.queryHandler(listQuery),
|
|
113
118
|
detail: r.queryHandler(detailQuery),
|
|
114
119
|
catalog: r.queryHandler(catalogQuery),
|
|
120
|
+
failures: r.queryHandler(tenantFailuresQuery),
|
|
115
121
|
};
|
|
116
122
|
|
|
117
123
|
const systemAdminAccess = { roles: ["SystemAdmin"] as const };
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
import { selectMany } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
2
|
+
import { access, defineQueryHandler } from "@cosmicdrift/kumiko-framework/engine";
|
|
3
|
+
import { InternalError } from "@cosmicdrift/kumiko-framework/errors";
|
|
4
|
+
import { parseJsonSafe } from "@cosmicdrift/kumiko-framework/utils";
|
|
5
|
+
import { z } from "zod";
|
|
6
|
+
import { tenantJobFailuresTable } from "../tenant-job-failure-table";
|
|
7
|
+
|
|
8
|
+
type TenantJobFailureRow = {
|
|
9
|
+
readonly tenantId: string;
|
|
10
|
+
readonly jobName: string;
|
|
11
|
+
readonly subject: string | null;
|
|
12
|
+
readonly messageKey: string;
|
|
13
|
+
readonly failedAt: Temporal.Instant;
|
|
14
|
+
};
|
|
15
|
+
|
|
16
|
+
const DEFAULT_LIMIT = 50;
|
|
17
|
+
|
|
18
|
+
// Mirrors job-runner.ts's jobSubjectKey: sorted field names, so a caller that
|
|
19
|
+
// passes the same subject values gets the same string the writer stored.
|
|
20
|
+
function subjectKey(subject: Record<string, string | number | boolean | null>): string {
|
|
21
|
+
return JSON.stringify(
|
|
22
|
+
Object.keys(subject)
|
|
23
|
+
.sort()
|
|
24
|
+
.map((field) => [field, subject[field] ?? null]),
|
|
25
|
+
);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
function isSubjectEntry(value: unknown): value is [string, unknown] {
|
|
29
|
+
return Array.isArray(value) && typeof value[0] === "string";
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function parseSubject(stored: string | null): Record<string, unknown> | null {
|
|
33
|
+
if (stored === null) return null;
|
|
34
|
+
// A corrupt subject must not fail the whole list — the record still tells
|
|
35
|
+
// the tenant which job failed and why. parseJsonSafe only survives a
|
|
36
|
+
// SyntaxError, so the shape needs its own check.
|
|
37
|
+
const parsed = parseJsonSafe<unknown>(stored, null);
|
|
38
|
+
if (!Array.isArray(parsed)) return null;
|
|
39
|
+
return Object.fromEntries(parsed.filter(isSubjectEntry));
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export const tenantFailuresQuery = defineQueryHandler({
|
|
43
|
+
name: "failures",
|
|
44
|
+
description:
|
|
45
|
+
"Lists the calling tenant's own failed jobs — one record per job and subject, newest first, each carrying a translation key for the reason, never the provider's own error message; use it to tell a tenant that their asynchronous job failed instead of leaving the screen waiting.",
|
|
46
|
+
schema: z.object({
|
|
47
|
+
jobName: z.string().optional(),
|
|
48
|
+
subject: z.record(z.string(), z.union([z.string(), z.number(), z.boolean()])).optional(),
|
|
49
|
+
limit: z.number().min(1).max(200).optional(),
|
|
50
|
+
}),
|
|
51
|
+
// Every membership rank: a failure record carries a job name and a
|
|
52
|
+
// translation key, nothing a team member of the tenant may not see.
|
|
53
|
+
access: { roles: access.roles("User", "Editor", ...access.admin) },
|
|
54
|
+
handler: async (query, ctx) => {
|
|
55
|
+
if (!ctx.systemDb) {
|
|
56
|
+
throw new InternalError({
|
|
57
|
+
message: "jobs:query:failures requires ctx.systemDb (feature must declare r.systemScope())",
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
// assertTenantMatch is a self-check on the caller and returns the same
|
|
61
|
+
// unfiltered system-mode db (tenant-db.ts) — the explicit tenantId below
|
|
62
|
+
// is the filter, assertRowsTenant the second net. There is no
|
|
63
|
+
// cross-tenant mode here: SystemAdmin reads runs via jobs:query:list.
|
|
64
|
+
const db = ctx.systemDb.assertTenantMatch(query.user.tenantId);
|
|
65
|
+
const where: Record<string, unknown> = { tenantId: [query.user.tenantId] };
|
|
66
|
+
if (query.payload.jobName) where["jobName"] = query.payload.jobName;
|
|
67
|
+
if (query.payload.subject) where["subject"] = subjectKey(query.payload.subject);
|
|
68
|
+
const rows = await selectMany<TenantJobFailureRow>(db, tenantJobFailuresTable, where, {
|
|
69
|
+
orderBy: { col: "failedAt", direction: "desc" },
|
|
70
|
+
limit: query.payload.limit ?? DEFAULT_LIMIT,
|
|
71
|
+
});
|
|
72
|
+
return {
|
|
73
|
+
rows: ctx.systemDb.assertRowsTenant(rows, "tenantId").map((row) => ({
|
|
74
|
+
jobName: row.jobName,
|
|
75
|
+
subject: parseSubject(row.subject),
|
|
76
|
+
messageKey: row.messageKey,
|
|
77
|
+
failedAt: row.failedAt,
|
|
78
|
+
})),
|
|
79
|
+
nextCursor: null,
|
|
80
|
+
};
|
|
81
|
+
},
|
|
82
|
+
});
|
package/src/jobs/index.ts
CHANGED
|
@@ -4,3 +4,4 @@ export type { JobRunLoggerCallbacks } from "./job-run-logger";
|
|
|
4
4
|
export { createJobRunLogger } from "./job-run-logger";
|
|
5
5
|
export type { JobLogLevel, JobRunStatus } from "./job-run-table";
|
|
6
6
|
export { jobRunLogsTable, jobRunsTable } from "./job-run-table";
|
|
7
|
+
export { tenantJobFailuresTable } from "./tenant-job-failure-table";
|
|
@@ -1,4 +1,10 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import {
|
|
2
|
+
deleteMany,
|
|
3
|
+
fetchOne,
|
|
4
|
+
insertMany,
|
|
5
|
+
insertOne,
|
|
6
|
+
updateMany,
|
|
7
|
+
} from "@cosmicdrift/kumiko-framework/bun-db";
|
|
2
8
|
import {
|
|
3
9
|
configuredPiiSubjectKms,
|
|
4
10
|
encryptPiiValueForSubject,
|
|
@@ -8,12 +14,18 @@ import {
|
|
|
8
14
|
} from "@cosmicdrift/kumiko-framework/crypto";
|
|
9
15
|
import type { DbConnection } from "@cosmicdrift/kumiko-framework/db";
|
|
10
16
|
import { type Registry, SYSTEM_TENANT_ID } from "@cosmicdrift/kumiko-framework/engine";
|
|
11
|
-
import type {
|
|
17
|
+
import type {
|
|
18
|
+
JobLogEntry,
|
|
19
|
+
JobMeta,
|
|
20
|
+
JobOutcomeMeta,
|
|
21
|
+
JobRunnerOptions,
|
|
22
|
+
} from "@cosmicdrift/kumiko-framework/jobs";
|
|
12
23
|
import { generateId } from "@cosmicdrift/kumiko-framework/utils";
|
|
13
24
|
import { mapWithConcurrency } from "../shared";
|
|
14
25
|
import { runCompletedSchema, runFailedSchema, runStartedSchema } from "./events";
|
|
15
26
|
import { parseJobInstant } from "./job-instant";
|
|
16
27
|
import { jobRunLogsTable, jobRunsTable } from "./job-run-table";
|
|
28
|
+
import { tenantJobFailuresTable } from "./tenant-job-failure-table";
|
|
17
29
|
|
|
18
30
|
// Matches PgKmsAdapter's default pool size (see tenant/handlers/*.query.ts) —
|
|
19
31
|
// bounds concurrent getOrCreateDek calls so a large log batch doesn't claim
|
|
@@ -138,6 +150,54 @@ async function encryptStartedPayload(
|
|
|
138
150
|
return encryptOrSentinel(kms, triggeredById, payload, "payload");
|
|
139
151
|
}
|
|
140
152
|
|
|
153
|
+
// fw#3079 — which row the tenant-visible failure record lives in: one per
|
|
154
|
+
// (tenant, job, subject), so the next outcome of the same work replaces or
|
|
155
|
+
// clears it. `subject` is null for a job that declares no subjectFields.
|
|
156
|
+
// Null target = nothing to write or clear: the job did not opt in, the run
|
|
157
|
+
// was tenant-less (cron resolves to SYSTEM_TENANT_ID, where no tenant-scoped
|
|
158
|
+
// query could ever read the row), or the caller predates fw#3079 and passes
|
|
159
|
+
// no outcome at all.
|
|
160
|
+
function tenantJobFailureTarget(
|
|
161
|
+
jobName: string,
|
|
162
|
+
outcome: JobOutcomeMeta | undefined,
|
|
163
|
+
): Record<string, unknown> | null {
|
|
164
|
+
if (!outcome?.tenantVisible || outcome.tenantId === SYSTEM_TENANT_ID) return null;
|
|
165
|
+
return { tenantId: outcome.tenantId, jobName, subject: outcome.tenantVisible.subject };
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
async function recordTenantJobFailure(
|
|
169
|
+
db: DbConnection,
|
|
170
|
+
jobName: string,
|
|
171
|
+
outcome: JobOutcomeMeta | undefined,
|
|
172
|
+
): Promise<void> {
|
|
173
|
+
const where = tenantJobFailureTarget(jobName, outcome);
|
|
174
|
+
const messageKey = outcome?.tenantVisible?.messageKey;
|
|
175
|
+
// skip: no tenant-visible target, or an attempt BullMQ may still retry — a
|
|
176
|
+
// non-final failure must not show the tenant a failure the next attempt
|
|
177
|
+
// may still resolve.
|
|
178
|
+
if (!where || !messageKey || outcome?.finalAttempt !== true) return;
|
|
179
|
+
// ponytail: delete-then-insert instead of an upsert — two runs of the same
|
|
180
|
+
// key finishing at once can leave two rows, and the query returns the
|
|
181
|
+
// newest. Add a unique index + ON CONFLICT if that ever matters.
|
|
182
|
+
await deleteMany(db, tenantJobFailuresTable, where);
|
|
183
|
+
await insertOne(db, tenantJobFailuresTable, {
|
|
184
|
+
...where,
|
|
185
|
+
messageKey,
|
|
186
|
+
failedAt: Temporal.Now.instant(),
|
|
187
|
+
});
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
async function clearTenantJobFailure(
|
|
191
|
+
db: DbConnection,
|
|
192
|
+
jobName: string,
|
|
193
|
+
outcome: JobOutcomeMeta | undefined,
|
|
194
|
+
): Promise<void> {
|
|
195
|
+
const where = tenantJobFailureTarget(jobName, outcome);
|
|
196
|
+
// skip: no tenant-visible target — nothing was ever recorded
|
|
197
|
+
if (!where) return;
|
|
198
|
+
await deleteMany(db, tenantJobFailuresTable, where);
|
|
199
|
+
}
|
|
200
|
+
|
|
141
201
|
export function createJobRunLogger(opts: JobRunLoggerOptions): JobRunLoggerCallbacks {
|
|
142
202
|
const { db } = opts;
|
|
143
203
|
|
|
@@ -241,11 +301,16 @@ export function createJobRunLogger(opts: JobRunLoggerOptions): JobRunLoggerCallb
|
|
|
241
301
|
},
|
|
242
302
|
|
|
243
303
|
onJobComplete: async (
|
|
244
|
-
|
|
304
|
+
jobName: string,
|
|
245
305
|
bullJobId: string,
|
|
246
306
|
duration: number,
|
|
247
307
|
logs: JobLogEntry[],
|
|
308
|
+
outcome?: JobOutcomeMeta,
|
|
248
309
|
) => {
|
|
310
|
+
// Before the run-row write and independent of it: a successful run
|
|
311
|
+
// clears the tenant's failure record even when the run row itself is
|
|
312
|
+
// unreachable (the state-loss return below).
|
|
313
|
+
await clearTenantJobFailure(db, jobName, outcome);
|
|
249
314
|
const resolved = await resolveRun(bullJobId);
|
|
250
315
|
// skip: state loss between start + complete (worker restart, cache
|
|
251
316
|
// evicted AND DB has no matching bull_job_id). Rare edge case; we
|
|
@@ -297,11 +362,15 @@ export function createJobRunLogger(opts: JobRunLoggerOptions): JobRunLoggerCallb
|
|
|
297
362
|
},
|
|
298
363
|
|
|
299
364
|
onJobFailed: async (
|
|
300
|
-
|
|
365
|
+
jobName: string,
|
|
301
366
|
bullJobId: string,
|
|
302
367
|
error: string,
|
|
303
368
|
logs: JobLogEntry[],
|
|
369
|
+
outcome?: JobOutcomeMeta,
|
|
304
370
|
) => {
|
|
371
|
+
// Mirror of onJobComplete: recorded independently of the run row, so a
|
|
372
|
+
// tenant still learns their job failed if the row is unreachable.
|
|
373
|
+
await recordTenantJobFailure(db, jobName, outcome);
|
|
305
374
|
const resolved = await resolveRun(bullJobId);
|
|
306
375
|
// skip: same rare state-loss case as in onJobComplete — drop the
|
|
307
376
|
// failure write rather than forge a run row from scratch.
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { asEntityTableMeta } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
2
|
+
import {
|
|
3
|
+
type EntityTableMeta,
|
|
4
|
+
instant,
|
|
5
|
+
table as pgTable,
|
|
6
|
+
serial,
|
|
7
|
+
sql,
|
|
8
|
+
text,
|
|
9
|
+
uuid,
|
|
10
|
+
} from "@cosmicdrift/kumiko-framework/db";
|
|
11
|
+
|
|
12
|
+
// The tenant's own view of a failed job (fw#3079). Direct-write store like
|
|
13
|
+
// store_job_runs: job-run-logger.ts writes it from the BullMQ callbacks,
|
|
14
|
+
// outside any dispatcher transaction.
|
|
15
|
+
//
|
|
16
|
+
// Deliberately holds no error text. `message_key` is a translation key
|
|
17
|
+
// (the thrown KumikoError's i18nKey or the one declared at the job), so a
|
|
18
|
+
// provider message can never reach the tenant through this table — it stays
|
|
19
|
+
// on store_job_runs.error and in store_job_run_logs, both SystemAdmin-only.
|
|
20
|
+
//
|
|
21
|
+
// One row per (tenant, job, subject): a later final-attempt failure replaces
|
|
22
|
+
// it, a later successful run of the same key deletes it. `subject` is the
|
|
23
|
+
// canonical JSON of the job's declared subjectFields, or NULL for a job that
|
|
24
|
+
// declares none — stored in clear, so a job must not declare a PII field
|
|
25
|
+
// as its subject.
|
|
26
|
+
export const tenantJobFailuresTable = pgTable("store_tenant_job_failures", {
|
|
27
|
+
id: serial("id").primaryKey(),
|
|
28
|
+
tenantId: uuid("tenant_id").notNull(),
|
|
29
|
+
jobName: text("job_name").notNull(),
|
|
30
|
+
subject: text("subject"),
|
|
31
|
+
messageKey: text("message_key").notNull(),
|
|
32
|
+
failedAt: instant("failed_at").default(sql`now()`).notNull(),
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
const derivedTenantJobFailuresTableMeta = asEntityTableMeta(tenantJobFailuresTable);
|
|
36
|
+
if (!derivedTenantJobFailuresTableMeta) {
|
|
37
|
+
throw new Error("tenantJobFailuresTable: table carries no EntityTableMeta — built via table()?");
|
|
38
|
+
}
|
|
39
|
+
export const tenantJobFailuresTableMeta: EntityTableMeta = derivedTenantJobFailuresTableMeta;
|
|
@@ -7,7 +7,12 @@
|
|
|
7
7
|
|
|
8
8
|
import { afterAll, beforeAll, beforeEach, describe, expect, test } from "bun:test";
|
|
9
9
|
import { defineFeature } from "@cosmicdrift/kumiko-framework/engine";
|
|
10
|
-
import {
|
|
10
|
+
import {
|
|
11
|
+
setupTestStack,
|
|
12
|
+
type TestStack,
|
|
13
|
+
TestUsers,
|
|
14
|
+
testTenantId,
|
|
15
|
+
} from "@cosmicdrift/kumiko-framework/stack";
|
|
11
16
|
import { z } from "zod";
|
|
12
17
|
import { createRateLimitingFeature } from "../feature";
|
|
13
18
|
|
|
@@ -93,3 +98,66 @@ describe("rate-limiting feature — status query", () => {
|
|
|
93
98
|
expect(res.status).toBe(403);
|
|
94
99
|
});
|
|
95
100
|
});
|
|
101
|
+
|
|
102
|
+
// The bucket key is caller-supplied, so the tenant boundary has to be drawn
|
|
103
|
+
// on it server-side. i18nKey is asserted alongside the code because a plain
|
|
104
|
+
// 403 could also come from tenant resolution and would pass either way.
|
|
105
|
+
const OUTSIDE_TENANT_KEY = "rateLimiting.errors.bucketOutsideTenant";
|
|
106
|
+
|
|
107
|
+
describe("rate-limiting feature — bucket tenant scope", () => {
|
|
108
|
+
test("denies an admin the bucket of another tenant", async () => {
|
|
109
|
+
const err = await stack.http.queryErr(
|
|
110
|
+
"rate-limiting:query:status",
|
|
111
|
+
{ bucket: `tenant:${testTenantId(2)}`, limit: 5, windowSeconds: 60 },
|
|
112
|
+
admin,
|
|
113
|
+
);
|
|
114
|
+
expect(err.code).toBe("access_denied");
|
|
115
|
+
expect(err.i18nKey).toBe(OUTSIDE_TENANT_KEY);
|
|
116
|
+
expect(err.httpStatus).toBe(403);
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
test("denies a key that only starts with the caller's tenant id", async () => {
|
|
120
|
+
const err = await stack.http.queryErr(
|
|
121
|
+
"rate-limiting:query:status",
|
|
122
|
+
{ bucket: `tenant:${admin.tenantId}-other`, limit: 5, windowSeconds: 60 },
|
|
123
|
+
admin,
|
|
124
|
+
);
|
|
125
|
+
expect(err.i18nKey).toBe(OUTSIDE_TENANT_KEY);
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
test("denies another user's bucket and the global IP buckets", async () => {
|
|
129
|
+
for (const bucket of [`user:${TestUsers.user.id}`, "l1:203.0.113.5", "ip:203.0.113.5"]) {
|
|
130
|
+
const err = await stack.http.queryErr(
|
|
131
|
+
"rate-limiting:query:status",
|
|
132
|
+
{ bucket, limit: 5, windowSeconds: 60 },
|
|
133
|
+
admin,
|
|
134
|
+
);
|
|
135
|
+
expect(err.i18nKey).toBe(OUTSIDE_TENANT_KEY);
|
|
136
|
+
}
|
|
137
|
+
});
|
|
138
|
+
|
|
139
|
+
test("allows the caller's own tenant and handler-scoped buckets", async () => {
|
|
140
|
+
for (const bucket of [
|
|
141
|
+
`tenant:${admin.tenantId}`,
|
|
142
|
+
`tenant+handler:${admin.tenantId}:rl-probe:query:ping`,
|
|
143
|
+
`user+handler:${admin.id}:rl-probe:query:ping`,
|
|
144
|
+
]) {
|
|
145
|
+
const status = await stack.http.queryOk<{ bucket: string; remaining: number }>(
|
|
146
|
+
"rate-limiting:query:status",
|
|
147
|
+
{ bucket, limit: 5, windowSeconds: 60 },
|
|
148
|
+
admin,
|
|
149
|
+
);
|
|
150
|
+
expect(status.bucket).toBe(bucket);
|
|
151
|
+
expect(status.remaining).toBe(5);
|
|
152
|
+
}
|
|
153
|
+
});
|
|
154
|
+
|
|
155
|
+
test("leaves SystemAdmin access to foreign buckets untouched", async () => {
|
|
156
|
+
const status = await stack.http.queryOk<{ bucket: string }>(
|
|
157
|
+
"rate-limiting:query:status",
|
|
158
|
+
{ bucket: `tenant:${testTenantId(2)}`, limit: 5, windowSeconds: 60 },
|
|
159
|
+
TestUsers.systemAdmin,
|
|
160
|
+
);
|
|
161
|
+
expect(status.bucket).toBe(`tenant:${testTenantId(2)}`);
|
|
162
|
+
});
|
|
163
|
+
});
|
|
@@ -1 +1,9 @@
|
|
|
1
|
-
[
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"version": "0.291.0",
|
|
4
|
+
"type": "breaking",
|
|
5
|
+
"title": "`rate-limiting:query:status` only peeks buckets of the calling tenant (fw#3076)",
|
|
6
|
+
"detail": "The handler took an arbitrary bucket key and peeked it, so a tenant Admin who knew or guessed another tenant's key could read that bucket's counter. The key is now matched segment-exact against the caller: `tenant:`/`tenant+handler:` must carry the caller's own tenant id, `user:`/`user+handler:` the caller's own user id. Everything else — other tenants, other users, the global `ip:`, `l1:` and `l2:` middleware buckets, malformed keys — is `access_denied` (`bucket_outside_tenant`) unless the caller is SystemAdmin, whose access is unchanged. The check runs before the resolver-wiring check, so a denied caller learns nothing about the mount either.",
|
|
7
|
+
"migration": "Only affects non-SystemAdmin callers of `rate-limiting:query:status`. A tenant\nAdmin keeps `tenant:<own>`, `tenant+handler:<own>:<handler>`, `user:<self>` and\n`user+handler:<self>:<handler>`. Two reads it had before now need SystemAdmin:\nanother user's bucket inside the same tenant (the tenant is not part of a\n`user:` key, so it cannot be verified without a membership lookup) and the\nglobal `ip:`/`l1:`/`l2:` buckets, which are not tenant-owned. Ops tooling that\npeeks those from a tenant Admin session has to run as SystemAdmin."
|
|
8
|
+
}
|
|
9
|
+
]
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { AccessDeniedError } from "@cosmicdrift/kumiko-framework/errors";
|
|
2
|
+
import { RateLimitErrors } from "../constants";
|
|
3
|
+
|
|
4
|
+
// Bucket keys are `<dimension>:<subject>[:<handler>]` (framework
|
|
5
|
+
// rate-limit/bucket.ts) — only `tenant*` and `user*` carry a subject the
|
|
6
|
+
// caller can own. `l1:`/`l2:` (middleware.ts) and every `ip*` bucket are
|
|
7
|
+
// global, so they stay SystemAdmin-only.
|
|
8
|
+
type BucketCaller = {
|
|
9
|
+
readonly id: string;
|
|
10
|
+
readonly tenantId: string;
|
|
11
|
+
readonly roles: readonly string[];
|
|
12
|
+
};
|
|
13
|
+
|
|
14
|
+
function ownsBucket(bucket: string, caller: BucketCaller): boolean {
|
|
15
|
+
const [dimension, subject, ...handler] = bucket.split(":");
|
|
16
|
+
if (dimension === undefined) return false;
|
|
17
|
+
const owner =
|
|
18
|
+
dimension === "tenant" || dimension === "tenant+handler"
|
|
19
|
+
? caller.tenantId
|
|
20
|
+
: dimension === "user" || dimension === "user+handler"
|
|
21
|
+
? caller.id
|
|
22
|
+
: undefined;
|
|
23
|
+
if (owner === undefined) return false;
|
|
24
|
+
// Segment-exact: a startsWith check would pass `tenant:<own-id>-other`.
|
|
25
|
+
if (subject !== owner) return false;
|
|
26
|
+
return dimension.endsWith("+handler") ? handler.length > 0 : handler.length === 0;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export function bucketAccessDenied(
|
|
30
|
+
bucket: string,
|
|
31
|
+
caller: BucketCaller,
|
|
32
|
+
): AccessDeniedError | undefined {
|
|
33
|
+
if (caller.roles.includes("SystemAdmin")) return undefined;
|
|
34
|
+
if (ownsBucket(bucket, caller)) return undefined;
|
|
35
|
+
return new AccessDeniedError({
|
|
36
|
+
i18nKey: "rateLimiting.errors.bucketOutsideTenant",
|
|
37
|
+
details: { reason: RateLimitErrors.bucketOutsideTenant },
|
|
38
|
+
});
|
|
39
|
+
}
|
|
@@ -2,6 +2,7 @@ import { defineQueryHandler } from "@cosmicdrift/kumiko-framework/engine";
|
|
|
2
2
|
import { UnprocessableError } from "@cosmicdrift/kumiko-framework/errors";
|
|
3
3
|
import { z } from "zod";
|
|
4
4
|
import { RateLimitErrors } from "../constants";
|
|
5
|
+
import { bucketAccessDenied } from "./bucket-access";
|
|
5
6
|
|
|
6
7
|
// Ops-side bucket inspection. Pass the bucket key (e.g. "user:42",
|
|
7
8
|
// "user+handler:42:orders:write:order:create") plus the limit/window the bucket
|
|
@@ -16,14 +17,15 @@ import { RateLimitErrors } from "../constants";
|
|
|
16
17
|
// Bucket key format is owned by the framework (see rate-limit/bucket.ts);
|
|
17
18
|
// callers pass the constructed key directly. We don't synthesize from
|
|
18
19
|
// (per, user, handler) here — peeking is a low-level op, the lookup
|
|
19
|
-
// surface stays small.
|
|
20
|
+
// surface stays small. Because the key is caller-supplied, bucket-access.ts
|
|
21
|
+
// restricts it to the caller's own tenant/user unless they are SystemAdmin.
|
|
20
22
|
export const rateLimitStatus = defineQueryHandler({
|
|
21
23
|
// Short name — the registry qualifies this to `rate-limiting:query:status`
|
|
22
24
|
// when the feature is registered. Passing the qualified form here would
|
|
23
25
|
// double-prefix it and the handler wouldn't be reachable.
|
|
24
26
|
name: "status",
|
|
25
27
|
description:
|
|
26
|
-
"Peeks at one rate-limit bucket and returns its remaining tokens, window and next reset without consuming a token; use it to explain why a caller is being throttled.",
|
|
28
|
+
"Peeks at one rate-limit bucket and returns its remaining tokens, window and next reset without consuming a token; use it to explain why a caller is being throttled. Non-SystemAdmin callers may only peek their own tenant's buckets (`tenant:`/`tenant+handler:`) and their own user buckets (`user:`/`user+handler:`); every other key, including the global `ip:`/`l1:`/`l2:` buckets, requires SystemAdmin.",
|
|
27
29
|
schema: z.object({
|
|
28
30
|
bucket: z.string().min(1),
|
|
29
31
|
limit: z.number().int().positive(),
|
|
@@ -31,6 +33,10 @@ export const rateLimitStatus = defineQueryHandler({
|
|
|
31
33
|
}),
|
|
32
34
|
access: { roles: ["Admin", "SystemAdmin"] },
|
|
33
35
|
handler: async (query, ctx) => {
|
|
36
|
+
// Before the resolver check, so a foreign-tenant caller learns nothing
|
|
37
|
+
// about the wiring either.
|
|
38
|
+
const denied = bucketAccessDenied(query.payload.bucket, ctx.user);
|
|
39
|
+
if (denied) throw denied;
|
|
34
40
|
if (!ctx.rateLimit) {
|
|
35
41
|
throw new UnprocessableError(RateLimitErrors.resolverUnavailable, {
|
|
36
42
|
i18nKey: "rateLimiting.errors.resolverUnavailable",
|