@vxil/cli 0.16.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config.d.ts +15 -0
- package/dist/vxil.js +50 -8
- package/package.json +3 -3
package/dist/config.d.ts
CHANGED
|
@@ -106,6 +106,21 @@ export interface CollectionDef {
|
|
|
106
106
|
* with a regression test on each — an attribute read by no enforcing path ships
|
|
107
107
|
* inert (the validation.unique / owner_field precedent). */
|
|
108
108
|
public?: boolean;
|
|
109
|
+
/** What a VERIFIED end user (a thin-client key + a session) may do on this
|
|
110
|
+
* collection (https://vxil.com/docs/guide/09-security-and-multitenancy):
|
|
111
|
+
* 'readwrite' (default) — read and write (owner-scoped when ownerField is set);
|
|
112
|
+
* 'read' — read only (owner-scoped as above); every end-user write
|
|
113
|
+
* (create, update, $inc, delete, publish, transaction step,
|
|
114
|
+
* filtered delete, batch write — and a delete elsewhere
|
|
115
|
+
* whose on_delete cascade / set_null would change a row
|
|
116
|
+
* here) is `403 server_only`;
|
|
117
|
+
* 'none' — no end-user reads or writes at all, even with ownerField.
|
|
118
|
+
* Server keys (and platform writes such as a jobs status mirror) are never
|
|
119
|
+
* affected. Stored on the collection (model data, not a config leaf) and
|
|
120
|
+
* carried by BOTH push paths; widening ('read'/'none' → 'readwrite') needs
|
|
121
|
+
* `vxil push --allow-destructive`. Any other value fails `vxil plan` /
|
|
122
|
+
* `vxil push` (it never falls back to 'readwrite'). */
|
|
123
|
+
endUserAccess?: 'readwrite' | 'read' | 'none';
|
|
109
124
|
/** Per-record action buttons (guide ch. 4): `[{ key, label, fn }]`, stored on
|
|
110
125
|
* the collection (model data, not a config leaf). The dashboard renders one
|
|
111
126
|
* button per action on each record row; `POST /v1/cms/items/:coll/:id/actions/
|
package/dist/vxil.js
CHANGED
|
@@ -4543,9 +4543,9 @@ function lowerAllTriggerBindings(def, name) {
|
|
|
4543
4543
|
return bindings;
|
|
4544
4544
|
}
|
|
4545
4545
|
var CONFIG_FILENAMES = ["vxil.config.ts", "vxil.config.mjs", "vxil.config.js"];
|
|
4546
|
-
var VXIL_CONFIG_PKG_VERSION = "0.
|
|
4547
|
-
var VXIL_SDK_PKG_VERSION = "0.
|
|
4548
|
-
var VXIL_CLI_PKG_VERSION = "0.
|
|
4546
|
+
var VXIL_CONFIG_PKG_VERSION = "0.11.0";
|
|
4547
|
+
var VXIL_SDK_PKG_VERSION = "0.17.0";
|
|
4548
|
+
var VXIL_CLI_PKG_VERSION = "0.17.0";
|
|
4549
4549
|
function ensureScaffoldPackageJson(cwd, opts = {}) {
|
|
4550
4550
|
const file = resolve(cwd, "package.json");
|
|
4551
4551
|
const wanted = {
|
|
@@ -8240,6 +8240,12 @@ function indexArmNote(rows) {
|
|
|
8240
8240
|
if (rows === null) return "arms inline or through the re-index push runs (live row count unavailable)";
|
|
8241
8241
|
return rows <= EQ_INDEX_INLINE_MAX_ROWS ? `arms inline (${rows} rows)` : `needs reindex (${rows} rows; push runs it)`;
|
|
8242
8242
|
}
|
|
8243
|
+
function endUserAccessOf(v) {
|
|
8244
|
+
return v === "read" || v === "none" ? v : "readwrite";
|
|
8245
|
+
}
|
|
8246
|
+
function endUserAccessRank(m) {
|
|
8247
|
+
return m === "none" ? 0 : m === "read" ? 1 : 2;
|
|
8248
|
+
}
|
|
8243
8249
|
function normalizeActions(raw) {
|
|
8244
8250
|
if (!Array.isArray(raw)) return [];
|
|
8245
8251
|
return raw.filter((a) => !!a && typeof a === "object").map((a) => ({ key: String(a.key), label: String(a.label), fn: String(a.fn) }));
|
|
@@ -8334,6 +8340,7 @@ async function readRemote2(api) {
|
|
|
8334
8340
|
collection: c.collection,
|
|
8335
8341
|
ownerField: c.owner_field ?? null,
|
|
8336
8342
|
public: c.public === true,
|
|
8343
|
+
endUserAccess: endUserAccessOf(c.end_user_access),
|
|
8337
8344
|
actions: normalizeActions(c.actions),
|
|
8338
8345
|
fields: Array.isArray(c.fields) ? c.fields : []
|
|
8339
8346
|
});
|
|
@@ -8382,6 +8389,12 @@ async function reindexCollection({ api, collection, field, onProgress }) {
|
|
|
8382
8389
|
return { scanned, updated, pages, skipped, ...indexCertify ? { indexCertify } : {} };
|
|
8383
8390
|
}
|
|
8384
8391
|
async function planCmsSchema({ api, collections, apply, allowDestructive = false, onProgress }) {
|
|
8392
|
+
for (const [name, def] of Object.entries(collections)) {
|
|
8393
|
+
const v = def.endUserAccess;
|
|
8394
|
+
if (v !== void 0 && v !== "readwrite" && v !== "read" && v !== "none") {
|
|
8395
|
+
throw new Error(`cms.collections.${name}.endUserAccess must be 'readwrite' | 'read' | 'none' (got ${JSON.stringify(v)})`);
|
|
8396
|
+
}
|
|
8397
|
+
}
|
|
8385
8398
|
const read = await readRemote2(api);
|
|
8386
8399
|
if (read.unavailable) {
|
|
8387
8400
|
if (apply) {
|
|
@@ -8526,6 +8539,23 @@ async function planCmsSchema({ api, collections, apply, allowDestructive = false
|
|
|
8526
8539
|
detail: `${name} public \u2192 ${declaredPublic} (keyless public delivery ${declaredPublic ? "ON" : "OFF"})`
|
|
8527
8540
|
});
|
|
8528
8541
|
}
|
|
8542
|
+
const declaredAccess = endUserAccessOf(def.endUserAccess);
|
|
8543
|
+
if (declaredAccess !== rc.endUserAccess) {
|
|
8544
|
+
const widens = endUserAccessRank(declaredAccess) > endUserAccessRank(rc.endUserAccess);
|
|
8545
|
+
if (!widens || allowDestructive) {
|
|
8546
|
+
collChanges.push({
|
|
8547
|
+
collection: name,
|
|
8548
|
+
kind: "set-end-user-access",
|
|
8549
|
+
detail: `${name} endUserAccess \u2192 '${declaredAccess}' (was '${rc.endUserAccess}')` + (widens ? " \u2014 WIDENS end-user access (full-config reconciliation)" : " (narrows end-user access)")
|
|
8550
|
+
});
|
|
8551
|
+
} else {
|
|
8552
|
+
collChanges.push({
|
|
8553
|
+
collection: name,
|
|
8554
|
+
kind: "warn",
|
|
8555
|
+
detail: `${name}: endUserAccess is '${rc.endUserAccess}' remotely but '${declaredAccess}' in config \u2014 that WIDENS what end users may do, so it is NOT applied. Declare endUserAccess: '${rc.endUserAccess}' to keep it, or re-run with --allow-destructive to widen.`
|
|
8556
|
+
});
|
|
8557
|
+
}
|
|
8558
|
+
}
|
|
8529
8559
|
const declaredActions = normalizeActions(def.actions);
|
|
8530
8560
|
if (stableStringify(declaredActions) !== stableStringify(rc.actions)) {
|
|
8531
8561
|
collChanges.push({
|
|
@@ -8597,6 +8627,8 @@ async function planCmsSchema({ api, collections, apply, allowDestructive = false
|
|
|
8597
8627
|
if (def.singular) body.singular = def.singular;
|
|
8598
8628
|
if (def.ownerField) body.owner_field = def.ownerField;
|
|
8599
8629
|
if (def.public === true) body.public = true;
|
|
8630
|
+
const access = endUserAccessOf(def.endUserAccess);
|
|
8631
|
+
if (access !== "readwrite") body.end_user_access = access;
|
|
8600
8632
|
if (def.actions && def.actions.length > 0) body.actions = normalizeActions(def.actions);
|
|
8601
8633
|
const res = await api("POST", "/v1/cms/collections", body);
|
|
8602
8634
|
if (res.status !== 201 && res.status !== 200 && res.body.error?.code !== "collection_exists") {
|
|
@@ -8636,6 +8668,14 @@ async function planCmsSchema({ api, collections, apply, allowDestructive = false
|
|
|
8636
8668
|
throw new Error(`cms: set public on ${c.collection} failed: ${e.code ?? res.status} ${e.message ?? ""}`);
|
|
8637
8669
|
}
|
|
8638
8670
|
applied++;
|
|
8671
|
+
} else if (c.kind === "set-end-user-access") {
|
|
8672
|
+
const target = endUserAccessOf(collections[c.collection].endUserAccess);
|
|
8673
|
+
const res = await api("POST", `/v1/cms/collections/${encodeURIComponent(c.collection)}/fields`, { end_user_access: target });
|
|
8674
|
+
if (res.status !== 200 && res.status !== 201) {
|
|
8675
|
+
const e = res.body.error ?? {};
|
|
8676
|
+
throw new Error(`cms: set endUserAccess on ${c.collection} failed: ${e.code ?? res.status} ${e.message ?? ""}`);
|
|
8677
|
+
}
|
|
8678
|
+
applied++;
|
|
8639
8679
|
} else if (c.kind === "set-actions") {
|
|
8640
8680
|
const target = normalizeActions(collections[c.collection].actions);
|
|
8641
8681
|
const res = await api("POST", `/v1/cms/collections/${encodeURIComponent(c.collection)}/fields`, { actions: target });
|
|
@@ -8701,7 +8741,7 @@ function formatCmsChanges(changes) {
|
|
|
8701
8741
|
if (c.kind === "destructive") return ` ! ${c.detail}`;
|
|
8702
8742
|
if (c.kind === "warn") return ` \u26A0 ${c.detail}`;
|
|
8703
8743
|
if (c.kind === "reindex") return ` \u21BB ${c.detail}`;
|
|
8704
|
-
if (c.kind === "alter-flags" || c.kind === "set-owner-field") return ` ~ ${c.detail}`;
|
|
8744
|
+
if (c.kind === "alter-flags" || c.kind === "set-owner-field" || c.kind === "set-end-user-access") return ` ~ ${c.detail}`;
|
|
8705
8745
|
return ` + ${c.detail}`;
|
|
8706
8746
|
}).join("\n");
|
|
8707
8747
|
}
|
|
@@ -8886,6 +8926,8 @@ var API_CHEATSHEET = [
|
|
|
8886
8926
|
{ feature: "jobs", what: "replay / cancel a run", cmd: "vxil api POST /v1/jobs/runs/<run_id>/replay \xB7 vxil api POST /v1/jobs/runs/<run_id>/cancel" },
|
|
8887
8927
|
{ feature: "jobs", what: "list schedules: held (fired, not started 2+ intervals), skipped_fires, overlap \u2014 fn-cron:* rows are function crons", cmd: "vxil api GET /v1/jobs/schedules" },
|
|
8888
8928
|
{ feature: "jobs", what: "create / pause / resume / delete a schedule", cmd: `vxil api POST /v1/jobs/schedules '{"job_name":"digest","target_url":"https://\u2026","cron":"0 8 * * *"}' \xB7 POST /v1/jobs/schedules/<schedule_id>/pause \xB7 POST \u2026/resume \xB7 vxil api DELETE /v1/jobs/schedules/<schedule_id>` },
|
|
8929
|
+
{ feature: "jobs", what: "edit a schedule in place (cron, timezone, target, payload)", cmd: `vxil api PATCH /v1/jobs/schedules/<schedule_id> '{"cron":"0 9 * * *","timezone":"Europe/Amman"}'` },
|
|
8930
|
+
{ feature: "jobs", what: "read a fan-in batch (counts by state, completed_at)", cmd: "vxil api GET /v1/jobs/batches/<batch_id>" },
|
|
8889
8931
|
{ feature: "jobs", what: "flow rules (rate + parallelism per endpoint or job)", cmd: `vxil api POST /v1/jobs/flow-rules '{"match_kind":"job_name","match_value":"resize","max_parallel":2}' \xB7 vxil api DELETE /v1/jobs/flow-rules/<rule_id>` },
|
|
8890
8932
|
// files
|
|
8891
8933
|
{ feature: "files", what: "upload a local file (mint \u2192 PUT the bytes \u2192 complete)", cmd: "vxil files put <path> --user <user_id> [--content-type <t>]" },
|
|
@@ -12091,7 +12133,7 @@ var TOOLS = [
|
|
|
12091
12133
|
{
|
|
12092
12134
|
name: "cms_list_collections",
|
|
12093
12135
|
feature: "cms",
|
|
12094
|
-
description: "List the tenant's content model: collections with their typed field definitions (the schema items are validated against). Each field row carries its attributes \u2014 required, validation, index_slot, relation_to, computed/compute, is_unique, on_delete, read_roles (the end-user org roles allowed to READ that field; null = ungated), and on an equality-indexed field indexed: true + index_ready (equality filters on it are index-served once ready; until then, and on unindexed unslotted fields, they are a bounded scan \u2014 same results). A field with read_roles is omitted from end-user reads without a matching role and is never served on the public lane, so a missing key in an item's data may be a read gate rather than an unset value.",
|
|
12136
|
+
description: "List the tenant's content model: collections with their typed field definitions (the schema items are validated against). Each field row carries its attributes \u2014 required, validation, index_slot, relation_to, computed/compute, is_unique, on_delete, read_roles (the end-user org roles allowed to READ that field; null = ungated), and on an equality-indexed field indexed: true + index_ready (equality filters on it are index-served once ready; until then, and on unindexed unslotted fields, they are a bounded scan \u2014 same results). A field with read_roles is omitted from end-user reads without a matching role and is never served on the public lane, so a missing key in an item's data may be a read gate rather than an unset value. Each collection row also carries end_user_access: 'readwrite' (default), 'read' (a verified end user \u2014 a thin-client key with a session \u2014 may read but every write answers 403 server_only) or 'none' (end-user reads and writes are 403 server_only); server keys are never affected.",
|
|
12095
12137
|
inputSchema: { type: "object", properties: {} },
|
|
12096
12138
|
method: "GET",
|
|
12097
12139
|
path: "/v1/cms/collections"
|
|
@@ -18467,7 +18509,7 @@ export default defineConfig({
|
|
|
18467
18509
|
"oidc_client_secret"
|
|
18468
18510
|
],
|
|
18469
18511
|
"configSrc": "import { defineConfig } from '@vxil/config';\n\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n// \"Ops dashboard\" \u2014 a PRIVATE dashboard for one team, wired to the team's own\n// services and stores: a payments provider, a shop or any REST API, the team's\n// other vxil projects, and inbound webhooks. Live KPI tiles, hourly and daily\n// trends, one updates feed, threshold alerts that open incidents, team roles.\n//\n// DISTRIBUTION OVER SHIPPED PRIMITIVES \u2014 there is no connectors feature: a\n// connector is a function + a secret + egressAllow. There is no dashboard\n// engine either: the tiles are cms rows, the trends are cms rows, the feed is\n// cms rows, and your own frontend renders them (examples/apps/team-dashboard).\n//\n// pull \u2192 functions `collect` polls each integration on a cron and writes\n// `metric_latest` (one row per series) + one `metric_points` row\n// per CLOSED hour; `drain-inbound` turns received webhooks into\n// `events_feed` rows\n// push \u2192 `drain-inbound` also binds `payments.` platform events, so the\n// payments integration's lifecycle lands in the feed in seconds\n// evaluate \u2192 `evaluate-alerts` checks `alert_rules` against `metric_latest`\n// and opens/resolves `incidents` once per crossing (+ chat + mail)\n// show \u2192 realtime change-data on three channels; staff read everything\n// with a read-only browser key; every staff WRITE goes through\n// `ops-action`, which checks the caller's org permission first\n//\n// Plan cadence: as shipped this is a Developer-plan config (4 crons, the\n// fastest every minute). The Free plan allows 5 functions and 3 crons, the\n// fastest every 15 minutes: set the two 5-minute schedules to '*/15 * * * *'\n// and drop drain-inbound's cron trigger (its `payments.` trigger still works),\n// or fold compact-daily into evaluate-alerts. See the README's limits table.\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n/** Org roles allowed to READ the restricted (revenue-grade) numbers. `owner` is\n * the built-in god role; `finance` and `ops-admin` are custom roles you define\n * once over the API (README step 3). An identity-provider group with the same\n * name works too: the IdP's `groups` claim and the org role travel together. */\nconst MONEY_READERS = ['finance', 'ops-admin', 'owner'];\n\nexport default defineConfig({\n env: 'staging',\n\n features: {\n // \u2500\u2500 Staff-only sign-in \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n // OIDC-ONLY + allowedDomains IS THE STAFF GATE. Never turn magic link or\n // email sign-up back on in a project holding company data: the collections\n // below are shared (no owner field) and every signed-in person reads all of\n // them, so an open sign-up would let ANY address read every number here.\n auth: {\n methods: { emailPassword: false, magicLink: false },\n providers: {\n oidc: {\n issuer: 'https://login.example-idp.com', // your identity provider (https, no query)\n clientId: 'ops-dashboard', // not a secret \u2014 it rides every authorize URL\n clientSecretRef: 'secret:oidc_client_secret',\n claims: { email: 'email', name: 'name', roles: 'groups' }, // IdP groups \u2192 session roles\n allowedDomains: ['example.com'], // fail-closed: any other domain is refused\n autoLink: true,\n },\n },\n // The member's org role rides the session, so `readRoles` gates without a\n // round-trip. It is a snapshot: revocation-grade checks call orgs (the\n // functions do, before every write).\n orgClaims: { enabled: true },\n session: { ttlMinutes: 60, refreshTtlDays: 7 },\n security: { allowedRedirectOrigins: ['https://ops.example.com', 'http://localhost:5173'] },\n },\n\n // Roles are rows, not config \u2014 the README defines ops-viewer, operator,\n // finance and ops-admin with `POST /v1/orgs/roles` and creates the one org\n // (slug `ops`) every staff member joins.\n orgs: { enabled: true },\n\n realtime: {},\n\n // One alert channel besides chat: an e-mail + in-app message to every org\n // member holding operator, ops-admin or owner, through the built-in\n // `transactional` template (subject / paragraph / call-to-action). The mock\n // provider delivers nothing; switch to your mail provider before relying on it.\n notifications: { provider: 'mock', fromEmail: 'ops@ops-dashboard.example', inboxEnabled: true },\n\n // Inbound webhook sources (one per external service that pushes to you)\n // are runtime rows: `POST /v1/webhooks/sources` \u2014 see the README.\n webhooks: {},\n\n functions: { enabled: true },\n\n cms: {\n // OFF on purpose: these collections have no owner \u2014 staff read all of\n // them \u2014 and the browser key holds NO cms:write, so an end-user session\n // can read but never write. Every write is a function.\n strictEndUserScope: false,\n // An hourly series is 8,760 rows a year; compact-daily prunes hour rows\n // after 90 days. Raise this if you track more than ~150 series.\n limits: { maxItemsPerCollection: 200_000 },\n\n hooks: {\n // The time bucket is derived from the point's own timestamp \u2014 the\n // writer cannot disagree with it. Hour: '2026-10-04T13'; day: '2026-10-04'.\n point_bucket: {\n collection: 'metric_points',\n event: 'beforeWrite',\n kind: 'derive',\n field: 'bucket',\n expr: \"item.grain == 'day' ? substr(item.ts, 0, 10) : substr(item.ts, 0, 13)\",\n },\n // One row per (series, grain, bucket): the unique point_key makes a\n // re-delivered cron tick a clean 409 instead of a duplicate point.\n // Derives run in declaration order, so this sees the bucket above.\n point_key: {\n collection: 'metric_points',\n event: 'beforeWrite',\n kind: 'derive',\n field: 'point_key',\n expr: \"concat(item.series, '|', item.grain, '|', item.bucket)\",\n },\n // Incidents move forward only: open \u2192 acked \u2192 resolved (or open \u2192 resolved).\n incident_stage: {\n collection: 'incidents',\n event: 'beforeUpdate',\n kind: 'validate',\n expr:\n 'item.state == before.state'\n + \" || (before.state == 'open' && (item.state == 'acked' || item.state == 'resolved'))\"\n + \" || (before.state == 'acked' && item.state == 'resolved')\",\n message: 'incidents move open \u2192 acked \u2192 resolved',\n },\n rule_op: {\n collection: 'alert_rules',\n event: 'beforeWrite',\n kind: 'validate',\n expr: \"item.op == '>' || item.op == '<' || item.op == '>=' || item.op == '<=' || item.op == 'stale'\",\n message: \"op must be one of > < >= <= stale\",\n },\n },\n\n // Realtime change-data. A FULL frame carries every field of the row to\n // every subscriber \u2014 including a `readRoles` field the subscriber may not\n // read \u2014 so a collection with any readRoles field publishes IDS ONLY and\n // the client re-reads the row (the read applies the gate). The CI gate\n // `cdc-gated-fields` refuses a full frame on such a collection.\n cdc: {\n tiles_live: { collection: 'metric_latest', channel: 'ops:metrics', events: ['created', 'updated'], payload: 'ids' },\n incidents_live: { collection: 'incidents', channel: 'ops:incidents', payload: 'ids' },\n feed_live: { collection: 'events_feed', channel: 'ops:events', events: ['created'], payload: 'full' },\n },\n\n // The DECLARATIVE twin of functions/compact-daily.ts, shown and left off:\n //\n // readModels: {\n // daily_by_series: {\n // collection: 'metric_points', kind: 'aggregate',\n // spec: {\n // aggregates: [{ fn: 'avg', field: 'value', as: 'avg' }, { fn: 'max', field: 'value', as: 'max' }, { fn: 'count' }],\n // groupBy: ['series', 'bucket'], filter: { grain: 'hour' }, window: { field: 'ts', sinceDays: 2 },\n // },\n // // materialize: { cron: '15 0 * * *', to: 'metric_daily' },\n // },\n // },\n //\n // A materialized read model rewrites a rollup collection on a schedule.\n // This blueprint keeps the imperative function instead because it also\n // compacts the gated restricted_value, writes day rows into the SAME\n // collection the charts already read, and prunes old hour rows.\n },\n },\n\n // \u2500\u2500 Schema-as-code (\u22648 index slots per collection: s1\u2013s4 / n1\u2013n2 / t1\u2013t2) \u2500\u2500\n cms: {\n collections: {\n // One row per connected source. NON-SECRET settings only \u2014 a token never\n // lives here (ops-action strips token-looking keys from every write).\n integrations: {\n singular: 'integration',\n fields: {\n key: { type: 'string', required: true, unique: true, indexSlot: 's1' }, // e.g. stripe, shop, billing-project\n kind: { type: 'string', required: true, indexSlot: 's2', validation: { enum: ['stripe', 'rest-json', 'vxil-project', 'webhook-source'] } },\n status: { type: 'string', indexSlot: 's3', validation: { enum: ['ok', 'stale', 'broken', 'could_not_check', 'paused'] } },\n owner: { type: 'string', indexSlot: 's4' }, // the person to ask\n consecutive_failures: { type: 'int', indexSlot: 'n1', validation: { min: 0 } },\n poll_every_min: { type: 'int', indexSlot: 'n2', validation: { min: 1, max: 1440 } },\n last_ok_at: { type: 'datetime', indexSlot: 't1' },\n next_due_at: { type: 'datetime', indexSlot: 't2' }, // collect reads `next_due_at <= now`, sorted\n label: { type: 'string', validation: { max: 120 } },\n enabled: { type: 'bool', indexed: true },\n // base_url, path, series[] (name, json_path, unit, sensitive), source_id,\n // aggregates[], link_url \u2014 never a credential\n config: { type: 'json' },\n last_error: { type: 'text', validation: { max: 500 } },\n last_error_at: { type: 'datetime' },\n cursor: { type: 'string' }, // webhook sources: the newest event already in the feed\n },\n },\n\n // The trend store: one row per series per CLOSED hour (grain 'hour') and\n // one per day (grain 'day', written by compact-daily). Never raw samples.\n metric_points: {\n singular: 'metric_point',\n fields: {\n series: { type: 'string', required: true, indexSlot: 's1' }, // 'source:metric[:dim]'\n grain: { type: 'string', required: true, indexSlot: 's2', validation: { enum: ['hour', 'day'] } },\n bucket: { type: 'string', indexSlot: 's3' }, // derived (point_bucket)\n point_key: { type: 'string', unique: true, indexSlot: 's4' }, // derived (point_key)\n value: { type: 'float', indexSlot: 'n1' },\n restricted_value: { type: 'float', indexSlot: 'n2', readRoles: MONEY_READERS },\n ts: { type: 'datetime', required: true, indexSlot: 't1' }, // the bucket's start\n unit: { type: 'string' },\n samples: { type: 'int' },\n min: { type: 'float' },\n max: { type: 'float' },\n },\n },\n\n // The tiles: one row per series, patched on every collect tick.\n metric_latest: {\n singular: 'metric_latest',\n fields: {\n series: { type: 'string', required: true, unique: true, indexSlot: 's1' },\n source: { type: 'string', indexSlot: 's2' }, // the integration key\n status: { type: 'string', indexSlot: 's3', validation: { enum: ['ok', 'warn', 'crit', 'stale'] } },\n last_bucket: { type: 'string', indexSlot: 's4' }, // the newest hour already in metric_points\n value: { type: 'float', indexSlot: 'n1' },\n restricted_value: { type: 'float', indexSlot: 'n2', readRoles: MONEY_READERS },\n updated_at: { type: 'datetime', indexSlot: 't1' },\n label: { type: 'string' },\n unit: { type: 'string' },\n sensitive: { type: 'bool' }, // true \u21D2 the number lives in restricted_value only\n hour_samples: { type: 'int' },\n hour_min: { type: 'float' }, // non-sensitive series only\n hour_max: { type: 'float' },\n },\n },\n\n // The updates feed. Title + summary only \u2014 never a vendor payload, which\n // can carry personal data, and this collection's change-data is full.\n events_feed: {\n singular: 'event',\n fields: {\n source: { type: 'string', required: true, indexSlot: 's1' },\n kind: { type: 'string', indexSlot: 's2' },\n severity: { type: 'string', indexSlot: 's3', validation: { enum: ['info', 'warn', 'error'] } },\n ext_id: { type: 'string', unique: true, indexSlot: 's4' }, // the sender's event id \u21D2 a redelivery is a 409\n occurred_at: { type: 'datetime', indexSlot: 't1' },\n title: { type: 'string', validation: { max: 200 } },\n url: { type: 'string', validation: { max: 500 } },\n summary: { type: 'text', validation: { max: 2000 } },\n },\n },\n\n alert_rules: {\n singular: 'alert_rule',\n fields: {\n key: { type: 'string', required: true, unique: true, indexSlot: 's1' },\n series: { type: 'string', required: true, indexSlot: 's2' },\n state: { type: 'string', indexSlot: 's3', validation: { enum: ['ok', 'firing'] } },\n severity: { type: 'string', indexSlot: 's4', validation: { enum: ['info', 'warn', 'crit'] } },\n threshold: { type: 'float', indexSlot: 'n1' }, // for 'stale': minutes without a fresh value\n for_minutes: { type: 'int', indexSlot: 'n2', validation: { min: 0 } },\n last_eval_at: { type: 'datetime', indexSlot: 't1' },\n last_fired_at: { type: 'datetime', indexSlot: 't2' },\n op: { type: 'string', required: true }, // > < >= <= stale (rule_op hook)\n enabled: { type: 'bool', indexed: true },\n channels: { type: 'json' }, // { chat: bool, email: bool }\n breach_since: { type: 'datetime' },\n repeat_after_hours: { type: 'int', validation: { min: 0 } },\n title: { type: 'string', validation: { max: 200 } },\n },\n },\n\n incidents: {\n singular: 'incident',\n fields: {\n rule: { type: 'string', required: true, indexSlot: 's1' },\n state: { type: 'string', required: true, indexSlot: 's2', validation: { enum: ['open', 'acked', 'resolved'] } },\n severity: { type: 'string', indexSlot: 's3' },\n acked_by: { type: 'string', indexSlot: 's4' },\n opened_at: { type: 'datetime', indexSlot: 't1' },\n resolved_at: { type: 'datetime', indexSlot: 't2' },\n title: { type: 'string', validation: { max: 200 } },\n last_value: { type: 'float' }, // null for a sensitive series\n note: { type: 'text', validation: { max: 2000 } },\n },\n },\n\n // Function memory: the `__compact__` row holds the last day compact-daily finished.\n ops_state: {\n singular: 'ops_state',\n fields: {\n key: { type: 'string', required: true, unique: true, indexSlot: 's1' },\n updated_at: { type: 'datetime', indexSlot: 't1' },\n cursor: { type: 'string' },\n },\n },\n },\n },\n\n // \u2500\u2500 The code that isn't config (guide ch. 8) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n functions: {\n // PULL. Every 5 minutes: poll each due integration, write the tiles and the\n // closed hour. Also an http door \u2014 the \"Sync now\" button \u2014 that checks the\n // caller holds `integrations.sync` in the ops org first.\n collect: {\n entry: './functions/collect.ts',\n trigger: { kind: 'cron', schedule: '*/5 * * * *' }, // Free: '*/15 * * * *'\n triggers: [{ kind: 'http' }],\n scopes: ['cms:read', 'cms:write', 'orgs:read'],\n // Every ref here must hold a value or the function refuses to run\n // (409 secret_missing): drop the sources you do not poll.\n secrets: ['secret:stripe_key', 'secret:shop_token', 'secret:vxil_peer_key'],\n // Deny-by-default egress: every REST host you poll goes here, then\n // `vxil push`. www.githubstatus.com is the keyless demo source. Which\n // host each SECRET may go to is fixed in functions/collect.ts\n // (SECRET_HOST / PEER_SECRETS), not by an integration row.\n egressAllow: ['api.stripe.com', 'shop.example.com', 'www.githubstatus.com'],\n limits: { timeoutMs: 20_000 }, // one slow vendor call fails in 20 s, not 30\n },\n\n // PUSH. Every minute: received webhooks \u2192 feed rows. And within seconds:\n // this project's own payments-integration events (the `payments.` trigger).\n 'drain-inbound': {\n entry: './functions/drain-inbound.ts',\n trigger: { kind: 'cron', schedule: '* * * * *' }, // Free: remove this trigger (3 crons max)\n triggers: [{ kind: 'webhook', source: 'payments.' }],\n scopes: ['cms:read', 'cms:write'],\n // Received webhook events are not on a function's scoped callback, so the\n // drain reads them with a key of THIS project holding only webhooks:read.\n secrets: ['secret:vxil_read_key'],\n egressAllow: [],\n },\n\n // EVALUATE. Every 5 minutes: rules \xD7 latest values \u2192 incidents, once per crossing.\n 'evaluate-alerts': {\n entry: './functions/evaluate-alerts.ts',\n trigger: { kind: 'cron', schedule: '*/5 * * * *' }, // Free: '*/15 * * * *'\n scopes: ['cms:read', 'cms:write', 'notifications:send', 'orgs:read'],\n secrets: ['secret:slack_webhook_url'],\n // The chat host. Add 'discord.com' for a Discord webhook or\n // 'chat.googleapis.com' for a Google Chat space webhook.\n egressAllow: ['hooks.slack.com'],\n },\n\n // Nightly: hour rows \u2192 one day row per series, then prune old hour rows.\n 'compact-daily': {\n entry: './functions/compact-daily.ts',\n trigger: { kind: 'cron', schedule: '15 0 * * *' },\n scopes: ['cms:read', 'cms:write'],\n egressAllow: [],\n },\n\n // SHOW (the write half). The dashboard's only write door: ack/resolve an\n // incident, edit or toggle a rule, edit or pause an integration \u2014 each\n // gated on an org permission. Audited as the project; the person is\n // recorded in the row (acked_by).\n 'ops-action': {\n entry: './functions/ops-action.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write', 'orgs:read'],\n egressAllow: [],\n },\n },\n\n // References only \u2014 values are set once with `vxil secrets set` and never appear here.\n secrets: {\n stripe_key: { feature: 'functions', description: 'Stripe RESTRICTED key, read-only on balance and charges' },\n shop_token: { feature: 'functions', description: 'bearer token of the REST API collect polls (rest-json integrations)' },\n vxil_peer_key: { feature: 'functions', description: 'API key of ANOTHER vxil project: usage:read (+ cms:read for aggregates)' },\n vxil_read_key: { feature: 'functions', description: 'API key of THIS project holding only webhooks:read' },\n slack_webhook_url: { feature: 'functions', description: 'Slack incoming webhook (or Discord / Google Chat space webhook) URL' },\n oidc_client_secret: { feature: 'auth', description: 'OIDC client secret of the staff identity provider' },\n },\n\n seed: {\n cms: [\n {\n collection: 'integrations',\n items: [\n {\n key: 'demo', kind: 'rest-json', label: 'Public status page (demo)', status: 'ok', enabled: true,\n poll_every_min: 5, consecutive_failures: 0, next_due_at: '2026-01-01T00:00:00Z',\n config: {\n base_url: 'https://www.githubstatus.com', path: '/api/v2/incidents/unresolved.json', auth: 'none',\n series: [{ name: 'open_incidents', json_path: 'incidents.length', unit: 'count' }],\n },\n },\n {\n key: 'stripe', kind: 'stripe', label: 'Payments provider', status: 'paused', enabled: false,\n poll_every_min: 5, consecutive_failures: 0, config: {},\n },\n {\n key: 'shop', kind: 'rest-json', label: 'Shop API', status: 'paused', enabled: false,\n poll_every_min: 15, consecutive_failures: 0,\n config: {\n base_url: 'https://shop.example.com', path: '/api/stats',\n series: [\n { name: 'orders_1h', json_path: 'orders.last_hour', unit: 'count' },\n { name: 'revenue_1h', json_path: 'revenue.last_hour', unit: 'usd', sensitive: true },\n ],\n },\n },\n {\n key: 'billing-project', kind: 'vxil-project', label: 'Billing backend (another vxil project)', status: 'paused',\n enabled: false, poll_every_min: 60, consecutive_failures: 0,\n config: { aggregates: [{ name: 'signups_24h', collection: 'accounts', fn: 'count', window_hours: 24 }] },\n },\n {\n key: 'support-inbox', kind: 'webhook-source', label: 'Support tool webhooks', status: 'paused', enabled: false,\n poll_every_min: 1, consecutive_failures: 0, config: { source_id: 'whs_replace_me' },\n },\n ],\n },\n {\n collection: 'alert_rules',\n items: [\n {\n key: 'payment-error-rate', series: 'stripe:charge_error_rate_1h', op: '>', threshold: 5, for_minutes: 10,\n severity: 'crit', state: 'ok', enabled: true, repeat_after_hours: 4, channels: { chat: true, email: true },\n title: 'Payment error rate above 5%',\n },\n {\n key: 'orders-dry', series: 'shop:orders_1h', op: '<', threshold: 1, for_minutes: 60,\n severity: 'warn', state: 'ok', enabled: true, repeat_after_hours: 24, channels: { chat: true, email: false },\n title: 'No orders in the last hour',\n },\n {\n key: 'demo-stale', series: 'demo:open_incidents', op: 'stale', threshold: 30, for_minutes: 0,\n severity: 'warn', state: 'ok', enabled: true, repeat_after_hours: 24, channels: { chat: true, email: false },\n title: 'Status data older than 30 minutes',\n },\n ],\n },\n ],\n },\n});\n",
|
|
18470
|
-
"readme": '# Ops dashboard (ops)\n\nA **private dashboard for one team**, wired to the team\'s own services and stores: your payments provider, a\nshop or any REST API, your other vxil projects, and the webhooks those services send you. It shows **live KPI\ntiles**, **hourly and daily trends**, **one updates feed**, **threshold alerts that open incidents** (once per\ncrossing, with a chat and an e-mail message), and **team roles** \u2014 revenue-grade numbers are readable only by the\nroles you name. Staff sign in through your identity provider and nothing else.\n\nIt is distribution over shipped building blocks. There is **no connectors feature**: a connector is a function +\na secret + a host in `egressAllow`. There is no dashboard engine either: every tile, point, feed entry and\nincident is a cms row, and your own frontend renders them \u2014 the reference frontend is\n`examples/apps/team-dashboard`.\n\nWhat it teaches that no other blueprint does: **time series on the cms** (closed-hour writes, a derived bucket,\nnightly compaction), **gated numbers** (`readRoles` + change-data that never carries them), and a **read-only\nbrowser** whose every write goes through one permission-checked function.\n\n## 1. The shape\n\n```\n your services \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500 vxil project (this blueprint) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500 your frontend\n (examples/apps/team-dashboard)\n payments provider \u2500\u2510 PULL \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510 metric_latest \u2500\u2500 tiles \u2500\u2500\u2510\n REST / shop API \u2500\u2500\u253C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u25B6\u2502 collect \u2502 metric_points \u2500\u2500 trends \u2500\u2524 \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510\n other vxil project \u2518 (cron) \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u251C\u2500\u2500\u25B6\u2502 read-only key \u2502\n \u25B2 compact-daily (nightly) \u2502 \u2502 + staff session \u2502\n inbound webhooks \u2500\u2500\u2510 PUSH \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510 \u2502 \u2502 + realtime \u2502\n payments events \u2500\u2500\u2534\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u25B6\u2502 drain-inbound\u2502 events_feed \u2500\u2500 feed \u2500\u2500\u2524 \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u252C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518\n \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u2502 \u2502 every write\n EVALUATE \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510 incidents \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u25BC\n alert_rules \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u25B6\u2502 evaluate-alerts\u2502\u2500\u2500\u25B6 chat + e-mail \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510\n \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u2502 ops-action \u2502 (org permission\n \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 checked first)\n```\n\nThe four arrows:\n\n| Arrow | Who | How |\n|---|---|---|\n| **Pull** | `collect`, every 5 minutes | polls each due integration, patches `metric_latest`, writes one `metric_points` row per closed hour |\n| **Push** | `drain-inbound`, every minute + on `payments.` events | received webhooks and the payments integration\'s events \u2192 `events_feed` |\n| **Evaluate** | `evaluate-alerts`, every 5 minutes | `alert_rules` \xD7 `metric_latest` \u2192 `incidents`, chat, e-mail \u2014 once per crossing |\n| **Show** | realtime change-data + your frontend | staff read every collection; the only write door is `ops-action` |\n\n## 2. Apply it\n\n```bash\nvxil init --template ops-dashboard\nvxil quickstart --env staging --no-push # or `vxil link <slug> --env staging` for an existing project\nvxil projects workload <slug> staging # Free: functions deploy to staging/development projects\n# the secrets (values never touch the config) \u2014 every ref a function declares must hold a value (below):\nprintf \'%s\' "$OIDC_SECRET" | vxil secrets set auth/oidc_client_secret\nprintf \'%s\' "$STRIPE_RK" | vxil secrets set functions/stripe_key # a RESTRICTED, read-only key\nprintf \'%s\' "$SHOP_TOKEN" | vxil secrets set functions/shop_token\nprintf \'%s\' "$PEER_KEY" | vxil secrets set functions/vxil_peer_key # a key of ANOTHER project\nprintf \'%s\' "$READ_KEY" | vxil secrets set functions/vxil_read_key # a key of THIS project: webhooks:read\nprintf \'%s\' "$SLACK_URL" | vxil secrets set functions/slack_webhook_url\nvxil push # collections, hooks, change-data, the five functions\nvxil seed # the demo integration + three example rules\nvxil functions invoke collect # one tick by hand: the demo source fills its first tile\n```\n\n**Every secret a function declares must hold a value**, or that function refuses to run\n(`409 secret_missing`, naming the ref). Before the push, delete from `collect`\'s `secrets` the sources you do\nnot poll (and their `egressAllow` hosts) \u2014 or store a placeholder value for a source you will add later.\n\nAlso replace the placeholders in `vxil.config.ts`: the OIDC `issuer`, `clientId` and\n`allowedDomains`, the `allowedRedirectOrigins` of your frontend, and `DASHBOARD_URL` in\n`functions/evaluate-alerts.ts`.\n\n| Secret | What it is | Least privilege |\n|---|---|---|\n| `oidc_client_secret` | your identity provider\'s client secret for this app | \u2014 |\n| `stripe_key` | a payments-provider **restricted** key | read on Balance and Charges, nothing else |\n| `shop_token` | the bearer token of the REST API a `rest-json` integration polls | a read-only token |\n| `vxil_peer_key` | an API key of **another** vxil project you own | `usage:read`, plus `cms:read` if you aggregate its collections |\n| `vxil_read_key` | an API key of **this** project | `webhooks:read` only \u2014 received events are not on a function\'s scoped callback |\n| `slack_webhook_url` | a chat incoming-webhook URL | \u2014 |\n\n`vxil_peer_key` and `vxil_read_key` are minted with `vxil keys mint --name <name> --scopes <a,b>` in the project\nthey belong to. `vxil keys mint` acts on your account, so it needs a dashboard session: run `vxil login` once\nfirst (a project key from `vxil quickstart` / `vxil link` is not enough). `vxil link --key` stores the key you give it in `~/.vxil/credentials.json` (so does `vxil login` with its session): treat that file as a secret \u2014 never commit it, copy it into an image or share it.\n\n## 3. Staff-only sign-in, and the roles\n\n`auth` here is **OIDC-only**: `methods.emailPassword` and `methods.magicLink` are off, and\n`providers.oidc.allowedDomains` lists your company domain. A sign-in whose verified e-mail is not on that list\nis refused (`403 oidc_domain_not_allowed`), and with no other method there is no other way in. **That is the\nstaff gate.** Do not turn magic link or e-mail sign-up back on in a project holding company data: these\ncollections have no owner field, every signed-in person reads all of them, so an open sign-up would let any\naddress on the internet read every number on the dashboard.\n\nRoles live in one org. Create it (slug `ops` \u2014 the functions look it up by slug) and the four custom roles once,\nwith a server key that holds `orgs:write`:\n\n```bash\nvxil api POST /v1/orgs --data \'{"slug":"ops","name":"Ops","owner_user_id":"<your user id>"}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"ops-viewer","name":"Viewer","permissions":["dashboard.read"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"operator","name":"Operator","permissions":["dashboard.read","incidents.manage","integrations.sync"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"finance","name":"Finance","permissions":["dashboard.read","revenue.read"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"ops-admin","name":"Ops admin","permissions":["dashboard.read","revenue.read","incidents.manage","integrations.sync","rules.manage","integrations.manage"]}\'\nvxil api POST /v1/orgs/org_\u2026/members --data \'{"user_id":"<user id>","role":"operator"}\'\n```\n\n`viewer` and `admin` are built-in role names with fixed permission sets, which is why the custom ones are\n`ops-viewer` and `ops-admin`. The built-in `owner` holds every permission. The permission strings are yours \u2014\nvxil only answers whether a member holds one (`GET /v1/orgs/{org}/check`).\n\nWith `orgClaims` on, the member\'s role rides the session, and the IdP\'s `groups` claim (named in\n`claims.roles`) rides beside it: a group called `finance` in your identity provider satisfies the same gate as\nthe org role.\n\n## 4. The browser keys\n\nMint two keys in the dashboard under **Keys \u2192 New key** (or with `vxil keys mint` after `vxil login`):\n\n| Key | Class | Scopes | Why |\n|---|---|---|---|\n| `web-auth` | Server | `auth:signin` | signing in is how a session is obtained, so this key cannot require one |\n| `web-data` | Public / thin-client | `cms:read`, `realtime:write`, `functions:invoke` | refused by the edge without a valid staff session |\n\n`web-data` holds **no `cms:write`**. A staff member can read every collection, mint a realtime connect token\nfor a channel (that is what `realtime:write` is for on a thin-client key \u2014 publishing over REST is refused to\nan end-user session) and call functions. Every write \u2014 acknowledging an incident, editing a rule, pausing an\nintegration \u2014 is `POST /v1/fn/ops-action`, which asks orgs whether this person holds the permission before it\nwrites. The write is audited as the project (the function writes with the project\'s cms scope, so the audit\nrow does not name the person); the human is recorded in the row itself \u2014 `acked_by` on an incident:\n\n```ts\nawait vx.asEndUser(session).fn[\'ops-action\']({ op: \'ack\', incident_id: \'itm_\u2026\', note: \'looking\' });\n// 403 { error: \'role_required\', permission: \'incidents.manage\' } for a viewer\n```\n\n| `op` | Permission | Body |\n|---|---|---|\n| `ack`, `resolve` | `incidents.manage` | `{ incident_id, note? }` \u2014 a compare-and-set on the state the button showed |\n| `upsert_rule` | `rules.manage` | `{ rule: { key, series, op, threshold, for_minutes?, severity?, repeat_after_hours?, channels?, title? } }` |\n| `toggle_rule` | `rules.manage` | `{ key, enabled }` |\n| `update_integration` | `integrations.manage` | `{ integration: { key, label?, owner?, poll_every_min?, config? } }` |\n| `pause_integration` | `integrations.manage` | `{ key, paused }` |\n\n"Sync now" is `POST /v1/fn/collect { integration: \'<key>\' }`, gated on `integrations.sync`.\n\n`update_integration` strips every config key that looks like a credential (`key`, `token`, `secret`,\n`password`, `authorization`, \u2026) and every value that looks like a bearer token, and names what it dropped. A\ncredential is a project secret, set with `vxil secrets set`, never from a browser.\n\nThe three settings that decide **where a credential is sent** \u2014 `config.base_url`, `config.auth` and\n`config.peer` \u2014 are not editable through `ops-action` at all: a change answers `409 locked_config` and an edit\nthat leaves them out keeps the stored values. They change only with a server key (`vxil api PATCH\n/v1/cms/items/integrations/<id>`, or a re-seed). `collect` enforces the same thing on its side, in code: each\nsecret is bound to one host (`SECRET_HOST` in `functions/collect.ts` \u2014 `shop_token` \u2192 your API\'s host;\n`stripe_key` \u2192 the payments provider\'s API; peer keys \u2192 the vxil API only, and only the names in\n`PEER_SECRETS`). A row that would send a credential anywhere else turns `broken` without a request being made.\n\n## 5. Adding a source\n\nA source is three edits: a **case in the `adapter` switch** in `functions/collect.ts`, the **host in\n`egressAllow`**, and a **secret**. Then `vxil push`. The egress guard is deny-by-default: a host you forgot\nanswers `403` and the integration turns `broken` with `egress blocked: add <host> to egressAllow and push`.\n\nBefore writing code, try the generic `rest-json` kind \u2014 a row, no code:\n\n```json\n{ "key": "shop", "kind": "rest-json", "enabled": true, "poll_every_min": 5,\n "next_due_at": "2026-01-01T00:00:00Z",\n "config": { "base_url": "https://shop.example.com", "path": "/api/stats",\n "series": [ { "name": "orders_1h", "json_path": "orders.last_hour", "unit": "count" },\n { "name": "revenue_1h", "json_path": "revenue.last_hour", "unit": "usd", "sensitive": true } ] } }\n```\n\nIt sends `Authorization: Bearer <shop_token>` (`"auth": "none"` for a public endpoint) and reads each\n`json_path` (`a.b.0.c`; a trailing `.length` counts an array). Series are named `<integration key>:<name>`.\nThe token goes **only** to the host in `SECRET_HOST.shop_token` at the top of `functions/collect.ts` \u2014 set it to\nyour API\'s host before the push; a `rest-json` row with a token pointing at any other host is refused. A second\nauthenticated REST API is a second secret and its own `case` (below), never a second row reusing `shop_token`.\n\n**A REST API with its own token** \u2014 one secret per vendor, one case:\n\n```ts\ncase \'helpdesk\': {\n const r = await getJson(\'https://api.helpdesk.example/v2/tickets/count?status=open\',\n { authorization: `Bearer ${secrets.helpdesk_token}` }) as { count: number };\n return [{ series: \'helpdesk:open_tickets\', value: r.count, unit: \'count\' }];\n}\n```\n\n**An analytics API that wants a service-account JWT** \u2014 sign it with WebCrypto inside the function (no SDK, no\nNode), then trade it for an access token:\n\n```ts\nasync function serviceToken(sa: { client_email: string; private_key: string }, scope: string): Promise<string> {\n const b64 = (s: string | ArrayBuffer) => btoa(typeof s === \'string\' ? s : String.fromCharCode(...new Uint8Array(s)))\n .replace(/=+$/, \'\').replace(/\\+/g, \'-\').replace(/\\//g, \'_\');\n const now = Math.floor(Date.now() / 1000);\n const unsigned = `${b64(JSON.stringify({ alg: \'RS256\', typ: \'JWT\' }))}.${b64(JSON.stringify({\n iss: sa.client_email, scope, aud: \'https://oauth2.example.com/token\', iat: now, exp: now + 3600 }))}`;\n const der = Uint8Array.from(atob(sa.private_key.replace(/-----[^-]+-----|\\s/g, \'\')), (c) => c.charCodeAt(0));\n const key = await crypto.subtle.importKey(\'pkcs8\', der, { name: \'RSASSA-PKCS1-v1_5\', hash: \'SHA-256\' }, false, [\'sign\']);\n const sig = await crypto.subtle.sign(\'RSASSA-PKCS1-v1_5\', key, new TextEncoder().encode(unsigned));\n const res = await fetch(\'https://oauth2.example.com/token\', { method: \'POST\',\n headers: { \'content-type\': \'application/x-www-form-urlencoded\' },\n body: `grant_type=urn:ietf:params:oauth:grant-type:jwt-bearer&assertion=${unsigned}.${b64(sig)}` });\n return ((await res.json()) as { access_token: string }).access_token;\n}\n```\n\nStore the service-account JSON as one secret, and add both the token host and the API host to `egressAllow`.\n\n**A SQL query over an HTTPS SQL endpoint** \u2014 the function is the SQL client; vxil never connects to your\ndatabase. The pattern, the credential handling and the least-privilege database user are in\n`examples/byo-db` (https://vxil.com/docs/guide/08-running-your-code-functions#your-own-database-from-a-function):\n\n```ts\ncase \'warehouse\': {\n const rows = await sql(secrets.database_url, \'select count(*)::int as n from orders where created_at > now() - interval \\\'1 hour\\\'\');\n return [{ series: \'warehouse:orders_1h\', value: rows[0].n, unit: \'count\' }];\n}\n```\n\n**A tool that pushes instead of being polled** \u2014 create an inbound webhook source and point the tool at its\nreceiver URL; then add an integration row of kind `webhook-source` with `config.source_id`:\n\n```ts\nconst { source_id, receiver_url_path } = await vx.webhooks.sources.create({ provider: \'generic\', name: \'support tool\' });\n// the receiver URL is returned once \u2014 paste it into the tool\'s webhook settings\n```\n\n## 6. Time series on the cms\n\n- **Write closed hours, not samples.** Every collect tick patches the series\' `metric_latest` row (value,\n `updated_at`, the hour\'s sample count and min/max). When a tick finds the stored sample belongs to an hour that\n has ended, it first writes that hour\'s closing sample as one `metric_points` row (`grain: \'hour\'`), then\n records the hour in `last_bucket`.\n- **The bucket is derived.** The `point_bucket` hook derives `bucket` from `ts` (`2026-10-04T13` for an hour,\n `2026-10-04` for a day) and `point_key` concatenates series, grain and bucket into a `unique` field \u2014 so a\n redelivered tick or two overlapping ticks produce a `409`, never a second point. Both are Lane-A `derive`\n hooks, not `validate`: the server computes the value on every write, so a writer that sends a wrong bucket (or\n none) is overwritten rather than refused, and no writer \u2014 `collect`, `compact-daily`, a backfill script \u2014 can\n file a point under the wrong hour. Keep them `derive` in any copy of this config.\n- **Compaction.** `compact-daily` aggregates yesterday\'s hour rows per series on the server (avg, min, max,\n count) into one `grain: \'day\'` row, then deletes hour rows older than 90 days and feed rows older than 30\n (the bounded filtered delete, 100 rows a call). Charts read `grain: \'hour\'` for the last days and\n `grain: \'day\'` beyond.\n- **The 50,000-row scan cap.** A server-side aggregate refuses (`422`) rather than scanning more than 50,000 rows.\n One day of hour rows is `series \xD7 24`, so compaction stays far below it even at 500 series; an aggregate over\n a year of raw 5-minute samples (105,120 per series) would not fit even for one series.\n- **Why not raw points.** Every cms write is an audited row and a change-data frame, and every call a function\n makes back into vxil counts as a request. A sample every 5 minutes is 288 rows a day per series; a closed hour\n is 24 \u2014 twelve times fewer writes, frames and requests, and twelve times fewer rows under every scan.\n\nThe config also shows (commented) the declarative twin of `compact-daily`, a `readModels` aggregate. It is left\noff because the function also compacts the gated value, writes into the collection the charts already read, and\nprunes.\n\n## 7. Sensitive numbers\n\nA series marked `sensitive` (revenue, balances) writes its number to `restricted_value`, which declares\n`readRoles: [\'finance\', \'ops-admin\', \'owner\']`; `value` stays null. A signed-in staff member without one of those\nroles gets the row with `restricted_value` **absent** and cannot filter, sort or group by it. Your server key\nand the functions (which run as the server) still see it, so alerts evaluate it \u2014 but a sensitive series\' number\nnever goes into a chat message, an e-mail or an incident row.\n\n**Change-data frames do not apply `readRoles`.** A `payload: \'full\'` frame carries the whole row to every\nsubscriber of the channel. So `metric_latest` and `incidents` publish **ids only** and the client re-reads the\nrow through the cms, where the gate applies; only `events_feed`, which has no gated field, publishes full rows.\nThe CI gate `cdc-gated-fields` refuses a full frame on any collection that declares a `readRoles` field.\n\n## 8. Alerts once per crossing\n\nA rule is `{ series, op, threshold, for_minutes, severity, repeat_after_hours, channels }`; `op` is `>`, `<`,\n`>=`, `<=` or `stale` (no fresh value for `threshold` minutes). `evaluate-alerts`:\n\n1. on a new breach stores `breach_since` and waits `for_minutes`;\n2. when the breach is sustained, creates the incident under `lock: \'incident:<rule>\'` with\n `guard: { filter: { rule, state: { $in: [\'open\', \'acked\'] } }, max: 1 }` \u2014 of two overlapping ticks exactly\n one gets a `201`, and only that one posts the chat message and sends the e-mail (the built-in `transactional`\n template, to every `ops` member holding `operator`, `ops-admin` or `owner`);\n3. stays quiet while it stays red, except one reminder every `repeat_after_hours`;\n4. if the incident create fails for any other reason (a `5xx`, a hook\'s `422`), leaves the rule `ok` with its\n `breach_since` and counts it in `failed` \u2014 the next tick crosses again, so the alert is delayed, never lost;\n5. on recovery flips the rule back (an `If-Match` on its version picks the one tick that does it), resolves the\n open incident and posts one RESOLVED message.\n\nThe chat poster sends Slack\'s `text` and Discord\'s `content`; for Discord add `discord.com` to `egressAllow`; a\nGoogle Chat space webhook (`chat.googleapis.com`, also added to `egressAllow`) gets `{ text }` alone.\n\n## 9. Reading your other vxil projects\n\nAn integration of kind `vxil-project` reads another project with **that project\'s own key** \u2014 one key per\nproject, minted there with the least it needs (`usage:read` for its request meter, `cms:read` to aggregate its\ncollections). It is pull, not push: the other project does not know this dashboard exists, and revoking the key\nthere disconnects it. A second project gets a second secret: declare `secret:<name>` on `collect`, add the name\nto `PEER_SECRETS` in `functions/collect.ts`, and point the row at it with `config.peer` (any name not on that\nlist is refused, so a row cannot pick some other secret and send it as a bearer key).\n\n```json\n{ "key": "billing-project", "kind": "vxil-project", "enabled": true, "poll_every_min": 60,\n "config": { "aggregates": [ { "name": "signups_24h", "collection": "accounts", "fn": "count", "window_hours": 24 } ] } }\n```\n\n## 10. Limits and plans\n\n| | Free (staging / development projects) | Developer |\n|---|---|---|\n| Functions / crons | 5 functions, 3 crons, fastest every 15 min | 50 functions, 20 crons, every minute |\n| As shipped | set the two `*/5` schedules (`collect`, `evaluate-alerts`) to `*/15`; drop `drain-inbound`\'s cron trigger (its `payments.` trigger stays) | runs as shipped |\n| Requests a month | 100,000 | 1,000,000 |\n\nEvery call a function makes back into vxil counts as a request (vendor calls do not). The arithmetic for 3\npolled integrations \xD7 5 series, a 30-day month:\n\n| Function | Per run | Runs a month | Requests |\n|---|---|---|---|\n| `collect` every 5 min | 1 list + per integration (1 list + 5 tile patches + 1 status patch) = 22 | 8,640 | ~190,000 |\n| hour closes | 15 series \xD7 1 point | 720 hours | ~11,000 |\n| `evaluate-alerts` every 5 min | 1 rules read + 1 latest read (+ writes on crossings only) | 8,640 | ~17,000 |\n| `drain-inbound` every minute, 1 source | 1 list + 1 events read (+ 1 write per event) | 43,200 | ~86,000 + events |\n| `compact-daily` | 2 aggregates + 15 day rows + retention deletes | 30 | ~1,500 |\n\nAbout **305,000 requests a month** on Developer. On Free, at 15-minute cadence with 2 integrations \xD7 4 series and\nno inbound drain: `collect` ~37,000, `evaluate-alerts` ~6,000, hour closes ~6,000 \u2014 about **50,000**. A tile list\nread by your frontend counts too, so let realtime change-data tell the page when to re-read instead of polling.\n\n## 11. Compose with\n\n- **alerts-to-slack** \u2014 alerts on the platform\'s own failure events (a dead-lettered job, a quarantined\n function, a failing payments webhook) next to this dashboard\'s business thresholds.\n- **service-monitor** \u2014 uptime checks for your endpoints, written into the same kind of rows.\n- **team-workspace** \u2014 the full org model: invitations, per-resource grants, a field only finance receives.\n\n## 12. Where the frontend lives\n\nThe **App builder** builds public-facing apps. An internal dashboard is your own frontend \u2014 any framework,\nany host \u2014 reading this project with the two keys above. The reference is `examples/apps/team-dashboard`:\nsign-in through your identity provider, tiles that re-read on change-data frames, trend charts over\n`metric_points`, the feed, and the incident buttons calling `ops-action`.\n',
|
|
18512
|
+
"readme": '# Ops dashboard (ops)\n\nA **private dashboard for one team**, wired to the team\'s own services and stores: your payments provider, a\nshop or any REST API, your other vxil projects, and the webhooks those services send you. It shows **live KPI\ntiles**, **hourly and daily trends**, **one updates feed**, **threshold alerts that open incidents** (once per\ncrossing, with a chat and an e-mail message), and **team roles** \u2014 revenue-grade numbers are readable only by the\nroles you name. Staff sign in through your identity provider and nothing else.\n\nIt is distribution over shipped building blocks. There is **no connectors feature**: a connector is a function +\na secret + a host in `egressAllow`. There is no dashboard engine either: every tile, point, feed entry and\nincident is a cms row, and your own frontend renders them \u2014 the reference frontend is\n`examples/apps/team-dashboard`.\n\nWhat it teaches that no other blueprint does: **time series on the cms** (closed-hour writes, a derived bucket,\nnightly compaction), **gated numbers** (`readRoles` + change-data that never carries them), and a **read-only\nbrowser** whose every write goes through one permission-checked function.\n\n## 1. The shape\n\n```\n your services \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500 vxil project (this blueprint) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500 your frontend\n (examples/apps/team-dashboard)\n payments provider \u2500\u2510 PULL \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510 metric_latest \u2500\u2500 tiles \u2500\u2500\u2510\n REST / shop API \u2500\u2500\u253C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u25B6\u2502 collect \u2502 metric_points \u2500\u2500 trends \u2500\u2524 \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510\n other vxil project \u2518 (cron) \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u251C\u2500\u2500\u25B6\u2502 read-only key \u2502\n \u25B2 compact-daily (nightly) \u2502 \u2502 + staff session \u2502\n inbound webhooks \u2500\u2500\u2510 PUSH \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510 \u2502 \u2502 + realtime \u2502\n payments events \u2500\u2500\u2534\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u25B6\u2502 drain-inbound\u2502 events_feed \u2500\u2500 feed \u2500\u2500\u2524 \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u252C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518\n \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u2502 \u2502 every write\n EVALUATE \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510 incidents \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u25BC\n alert_rules \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u25B6\u2502 evaluate-alerts\u2502\u2500\u2500\u25B6 chat + e-mail \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510\n \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u2502 ops-action \u2502 (org permission\n \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 checked first)\n```\n\nThe four arrows:\n\n| Arrow | Who | How |\n|---|---|---|\n| **Pull** | `collect`, every 5 minutes | polls each due integration, patches `metric_latest`, writes one `metric_points` row per closed hour |\n| **Push** | `drain-inbound`, every minute + on `payments.` events | received webhooks and the payments integration\'s events \u2192 `events_feed` |\n| **Evaluate** | `evaluate-alerts`, every 5 minutes | `alert_rules` \xD7 `metric_latest` \u2192 `incidents`, chat, e-mail \u2014 once per crossing |\n| **Show** | realtime change-data + your frontend | staff read every collection; the only write door is `ops-action` |\n\n## 2. Apply it\n\n```bash\nmkdir ops-dashboard && cd ops-dashboard && vxil init --template ops-dashboard # init scaffolds into the current directory\nvxil quickstart --env staging --no-push # or `vxil link <slug> --env staging` for an existing project\nvxil projects workload <slug> staging # Free: functions deploy to staging/development projects\n# the secrets (values never touch the config) \u2014 every ref a function declares must hold a value (below):\nprintf \'%s\' "$OIDC_SECRET" | vxil secrets set auth/oidc_client_secret\nprintf \'%s\' "$STRIPE_RK" | vxil secrets set functions/stripe_key # a RESTRICTED, read-only key\nprintf \'%s\' "$SHOP_TOKEN" | vxil secrets set functions/shop_token\nprintf \'%s\' "$PEER_KEY" | vxil secrets set functions/vxil_peer_key # a key of ANOTHER project\nprintf \'%s\' "$READ_KEY" | vxil secrets set functions/vxil_read_key # a key of THIS project: webhooks:read\nprintf \'%s\' "$SLACK_URL" | vxil secrets set functions/slack_webhook_url\nvxil push # collections, hooks, change-data, the five functions\nvxil seed # the demo integration + three example rules\nvxil functions invoke collect # one tick by hand: the demo source fills its first tile\n```\n\n**Every secret a function declares must hold a value**, or that function refuses to run\n(`409 secret_missing`, naming the ref). Before the push, delete from `collect`\'s `secrets` the sources you do\nnot poll (and their `egressAllow` hosts) \u2014 or store a placeholder value for a source you will add later.\n\nAlso replace the placeholders in `vxil.config.ts`: the OIDC `issuer`, `clientId` and\n`allowedDomains`, the `allowedRedirectOrigins` of your frontend, and `DASHBOARD_URL` in\n`functions/evaluate-alerts.ts`.\n\n| Secret | What it is | Least privilege |\n|---|---|---|\n| `oidc_client_secret` | your identity provider\'s client secret for this app | \u2014 |\n| `stripe_key` | a payments-provider **restricted** key | read on Balance and Charges, nothing else |\n| `shop_token` | the bearer token of the REST API a `rest-json` integration polls | a read-only token |\n| `vxil_peer_key` | an API key of **another** vxil project you own | `usage:read`, plus `cms:read` if you aggregate its collections |\n| `vxil_read_key` | an API key of **this** project | `webhooks:read` only \u2014 received events are not on a function\'s scoped callback |\n| `slack_webhook_url` | a chat incoming-webhook URL | \u2014 |\n\n`vxil_peer_key` and `vxil_read_key` are minted with `vxil keys mint --name <name> --scopes <a,b>` in the project\nthey belong to. `vxil keys mint` acts on your account, so it needs a dashboard session: run `vxil login` once\nfirst (a project key from `vxil quickstart` / `vxil link` is not enough). `vxil link --key` stores the key you give it in `~/.vxil/credentials.json` (so does `vxil login` with its session): treat that file as a secret \u2014 never commit it, copy it into an image or share it.\n\n## 3. Staff-only sign-in, and the roles\n\n`auth` here is **OIDC-only**: `methods.emailPassword` and `methods.magicLink` are off, and\n`providers.oidc.allowedDomains` lists your company domain. A sign-in whose verified e-mail is not on that list\nis refused (`403 oidc_domain_not_allowed`), and with no other method there is no other way in. **That is the\nstaff gate.** Do not turn magic link or e-mail sign-up back on in a project holding company data: these\ncollections have no owner field, every signed-in person reads all of them, so an open sign-up would let any\naddress on the internet read every number on the dashboard.\n\nRoles live in one org. Create it (slug `ops` \u2014 the functions look it up by slug) and the four custom roles once,\nwith a server key that holds `orgs:write`:\n\n```bash\nvxil api POST /v1/orgs --data \'{"slug":"ops","name":"Ops","owner_user_id":"<your user id>"}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"ops-viewer","name":"Viewer","permissions":["dashboard.read"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"operator","name":"Operator","permissions":["dashboard.read","incidents.manage","integrations.sync"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"finance","name":"Finance","permissions":["dashboard.read","revenue.read"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"ops-admin","name":"Ops admin","permissions":["dashboard.read","revenue.read","incidents.manage","integrations.sync","rules.manage","integrations.manage"]}\'\nvxil api POST /v1/orgs/org_\u2026/members --data \'{"user_id":"<user id>","role":"operator"}\'\n```\n\n`viewer` and `admin` are built-in role names with fixed permission sets, which is why the custom ones are\n`ops-viewer` and `ops-admin`. The built-in `owner` holds every permission. The permission strings are yours \u2014\nvxil only answers whether a member holds one (`GET /v1/orgs/{org}/check`).\n\nWith `orgClaims` on, the member\'s role rides the session, and the IdP\'s `groups` claim (named in\n`claims.roles`) rides beside it: a group called `finance` in your identity provider satisfies the same gate as\nthe org role.\n\n## 4. The browser keys\n\nMint two keys in the dashboard under **Keys \u2192 New key** (or with `vxil keys mint` after `vxil login`):\n\n| Key | Class | Scopes | Why |\n|---|---|---|---|\n| `web-auth` | Server | `auth:signin` | signing in is how a session is obtained, so this key cannot require one |\n| `web-data` | Public / thin-client | `cms:read`, `realtime:write`, `functions:invoke` | refused by the edge without a valid staff session |\n\n`web-data` holds **no `cms:write`**. A staff member can read every collection, mint a realtime connect token\nfor a channel (that is what `realtime:write` is for on a thin-client key \u2014 publishing over REST is refused to\nan end-user session) and call functions. Every write \u2014 acknowledging an incident, editing a rule, pausing an\nintegration \u2014 is `POST /v1/fn/ops-action`, which asks orgs whether this person holds the permission before it\nwrites. The write is audited as the project (the function writes with the project\'s cms scope, so the audit\nrow does not name the person); the human is recorded in the row itself \u2014 `acked_by` on an incident:\n\n```ts\nawait vx.asEndUser(session).fn[\'ops-action\']({ op: \'ack\', incident_id: \'itm_\u2026\', note: \'looking\' });\n// 403 { error: \'role_required\', permission: \'incidents.manage\' } for a viewer\n```\n\n| `op` | Permission | Body |\n|---|---|---|\n| `ack`, `resolve` | `incidents.manage` | `{ incident_id, note? }` \u2014 a compare-and-set on the state the button showed |\n| `upsert_rule` | `rules.manage` | `{ rule: { key, series, op, threshold, for_minutes?, severity?, repeat_after_hours?, channels?, title? } }` |\n| `toggle_rule` | `rules.manage` | `{ key, enabled }` |\n| `update_integration` | `integrations.manage` | `{ integration: { key, label?, owner?, poll_every_min?, config? } }` |\n| `pause_integration` | `integrations.manage` | `{ key, paused }` |\n\n"Sync now" is `POST /v1/fn/collect { integration: \'<key>\' }`, gated on `integrations.sync`.\n\n`update_integration` strips every config key that looks like a credential (`key`, `token`, `secret`,\n`password`, `authorization`, \u2026) and every value that looks like a bearer token, and names what it dropped. A\ncredential is a project secret, set with `vxil secrets set`, never from a browser.\n\nThe three settings that decide **where a credential is sent** \u2014 `config.base_url`, `config.auth` and\n`config.peer` \u2014 are not editable through `ops-action` at all: a change answers `409 locked_config` and an edit\nthat leaves them out keeps the stored values. They change only with a server key (`vxil api PATCH\n/v1/cms/items/integrations/<id>`, or a re-seed). `collect` enforces the same thing on its side, in code: each\nsecret is bound to one host (`SECRET_HOST` in `functions/collect.ts` \u2014 `shop_token` \u2192 your API\'s host;\n`stripe_key` \u2192 the payments provider\'s API; peer keys \u2192 the vxil API only, and only the names in\n`PEER_SECRETS`). A row that would send a credential anywhere else turns `broken` without a request being made.\n\n## 5. Adding a source\n\nA source is three edits: a **case in the `adapter` switch** in `functions/collect.ts`, the **host in\n`egressAllow`**, and a **secret**. Then `vxil push`. The egress guard is deny-by-default: a host you forgot\nanswers `403` and the integration turns `broken` with `egress blocked: add <host> to egressAllow and push`.\n\nBefore writing code, try the generic `rest-json` kind \u2014 a row, no code:\n\n```json\n{ "key": "shop", "kind": "rest-json", "enabled": true, "poll_every_min": 5,\n "next_due_at": "2026-01-01T00:00:00Z",\n "config": { "base_url": "https://shop.example.com", "path": "/api/stats",\n "series": [ { "name": "orders_1h", "json_path": "orders.last_hour", "unit": "count" },\n { "name": "revenue_1h", "json_path": "revenue.last_hour", "unit": "usd", "sensitive": true } ] } }\n```\n\nIt sends `Authorization: Bearer <shop_token>` (`"auth": "none"` for a public endpoint) and reads each\n`json_path` (`a.b.0.c`; a trailing `.length` counts an array). Series are named `<integration key>:<name>`.\nThe token goes **only** to the host in `SECRET_HOST.shop_token` at the top of `functions/collect.ts` \u2014 set it to\nyour API\'s host before the push; a `rest-json` row with a token pointing at any other host is refused. A second\nauthenticated REST API is a second secret and its own `case` (below), never a second row reusing `shop_token`.\n\n**A REST API with its own token** \u2014 one secret per vendor, one case:\n\n```ts\ncase \'helpdesk\': {\n const r = await getJson(\'https://api.helpdesk.example/v2/tickets/count?status=open\',\n { authorization: `Bearer ${secrets.helpdesk_token}` }) as { count: number };\n return [{ series: \'helpdesk:open_tickets\', value: r.count, unit: \'count\' }];\n}\n```\n\n**An analytics API that wants a service-account JWT** \u2014 sign it with WebCrypto inside the function (no SDK, no\nNode), then trade it for an access token:\n\n```ts\nasync function serviceToken(sa: { client_email: string; private_key: string }, scope: string): Promise<string> {\n const b64 = (s: string | ArrayBuffer) => btoa(typeof s === \'string\' ? s : String.fromCharCode(...new Uint8Array(s)))\n .replace(/=+$/, \'\').replace(/\\+/g, \'-\').replace(/\\//g, \'_\');\n const now = Math.floor(Date.now() / 1000);\n const unsigned = `${b64(JSON.stringify({ alg: \'RS256\', typ: \'JWT\' }))}.${b64(JSON.stringify({\n iss: sa.client_email, scope, aud: \'https://oauth2.example.com/token\', iat: now, exp: now + 3600 }))}`;\n const der = Uint8Array.from(atob(sa.private_key.replace(/-----[^-]+-----|\\s/g, \'\')), (c) => c.charCodeAt(0));\n const key = await crypto.subtle.importKey(\'pkcs8\', der, { name: \'RSASSA-PKCS1-v1_5\', hash: \'SHA-256\' }, false, [\'sign\']);\n const sig = await crypto.subtle.sign(\'RSASSA-PKCS1-v1_5\', key, new TextEncoder().encode(unsigned));\n const res = await fetch(\'https://oauth2.example.com/token\', { method: \'POST\',\n headers: { \'content-type\': \'application/x-www-form-urlencoded\' },\n body: `grant_type=urn:ietf:params:oauth:grant-type:jwt-bearer&assertion=${unsigned}.${b64(sig)}` });\n return ((await res.json()) as { access_token: string }).access_token;\n}\n```\n\nStore the service-account JSON as one secret, and add both the token host and the API host to `egressAllow`.\n\n**A SQL query over an HTTPS SQL endpoint** \u2014 the function is the SQL client; vxil never connects to your\ndatabase. The pattern, the credential handling and the least-privilege database user are in\n`examples/byo-db` (https://vxil.com/docs/guide/08-running-your-code-functions#your-own-database-from-a-function):\n\n```ts\ncase \'warehouse\': {\n const rows = await sql(secrets.database_url, \'select count(*)::int as n from orders where created_at > now() - interval \\\'1 hour\\\'\');\n return [{ series: \'warehouse:orders_1h\', value: rows[0].n, unit: \'count\' }];\n}\n```\n\n**A tool that pushes instead of being polled** \u2014 create an inbound webhook source and point the tool at its\nreceiver URL; then add an integration row of kind `webhook-source` with `config.source_id`:\n\n```ts\nconst { source_id, receiver_url_path } = await vx.webhooks.sources.create({ provider: \'generic\', name: \'support tool\' });\n// the receiver URL is returned once \u2014 paste it into the tool\'s webhook settings\n```\n\n## 6. Time series on the cms\n\n- **Write closed hours, not samples.** Every collect tick patches the series\' `metric_latest` row (value,\n `updated_at`, the hour\'s sample count and min/max). When a tick finds the stored sample belongs to an hour that\n has ended, it first writes that hour\'s closing sample as one `metric_points` row (`grain: \'hour\'`), then\n records the hour in `last_bucket`.\n- **The bucket is derived.** The `point_bucket` hook derives `bucket` from `ts` (`2026-10-04T13` for an hour,\n `2026-10-04` for a day) and `point_key` concatenates series, grain and bucket into a `unique` field \u2014 so a\n redelivered tick or two overlapping ticks produce a `409`, never a second point. Both are Lane-A `derive`\n hooks, not `validate`: the server computes the value on every write, so a writer that sends a wrong bucket (or\n none) is overwritten rather than refused, and no writer \u2014 `collect`, `compact-daily`, a backfill script \u2014 can\n file a point under the wrong hour. Keep them `derive` in any copy of this config.\n- **Compaction.** `compact-daily` aggregates yesterday\'s hour rows per series on the server (avg, min, max,\n count) into one `grain: \'day\'` row, then deletes hour rows older than 90 days and feed rows older than 30\n (the bounded filtered delete, 100 rows a call). Charts read `grain: \'hour\'` for the last days and\n `grain: \'day\'` beyond.\n- **The 50,000-row scan cap.** A server-side aggregate refuses (`422`) rather than scanning more than 50,000 rows.\n One day of hour rows is `series \xD7 24`, so compaction stays far below it even at 500 series; an aggregate over\n a year of raw 5-minute samples (105,120 per series) would not fit even for one series.\n- **Why not raw points.** Every cms write is an audited row and a change-data frame, and every call a function\n makes back into vxil counts as a request. A sample every 5 minutes is 288 rows a day per series; a closed hour\n is 24 \u2014 twelve times fewer writes, frames and requests, and twelve times fewer rows under every scan.\n\nThe config also shows (commented) the declarative twin of `compact-daily`, a `readModels` aggregate. It is left\noff because the function also compacts the gated value, writes into the collection the charts already read, and\nprunes.\n\n## 7. Sensitive numbers\n\nA series marked `sensitive` (revenue, balances) writes its number to `restricted_value`, which declares\n`readRoles: [\'finance\', \'ops-admin\', \'owner\']`; `value` stays null. A signed-in staff member without one of those\nroles gets the row with `restricted_value` **absent** and cannot filter, sort or group by it. Your server key\nand the functions (which run as the server) still see it, so alerts evaluate it \u2014 but a sensitive series\' number\nnever goes into a chat message, an e-mail or an incident row.\n\n**Change-data frames do not apply `readRoles`.** A `payload: \'full\'` frame carries the whole row to every\nsubscriber of the channel. So `metric_latest` and `incidents` publish **ids only** and the client re-reads the\nrow through the cms, where the gate applies; only `events_feed`, which has no gated field, publishes full rows.\nThe CI gate `cdc-gated-fields` refuses a full frame on any collection that declares a `readRoles` field.\n\n## 8. Alerts once per crossing\n\nA rule is `{ series, op, threshold, for_minutes, severity, repeat_after_hours, channels }`; `op` is `>`, `<`,\n`>=`, `<=` or `stale` (no fresh value for `threshold` minutes). `evaluate-alerts`:\n\n1. on a new breach stores `breach_since` and waits `for_minutes`;\n2. when the breach is sustained, creates the incident under `lock: \'incident:<rule>\'` with\n `guard: { filter: { rule, state: { $in: [\'open\', \'acked\'] } }, max: 1 }` \u2014 of two overlapping ticks exactly\n one gets a `201`, and only that one posts the chat message and sends the e-mail (the built-in `transactional`\n template, to every `ops` member holding `operator`, `ops-admin` or `owner`);\n3. stays quiet while it stays red, except one reminder every `repeat_after_hours`;\n4. if the incident create fails for any other reason (a `5xx`, a hook\'s `422`), leaves the rule `ok` with its\n `breach_since` and counts it in `failed` \u2014 the next tick crosses again, so the alert is delayed, never lost;\n5. on recovery flips the rule back (an `If-Match` on its version picks the one tick that does it), resolves the\n open incident and posts one RESOLVED message.\n\nThe chat poster sends Slack\'s `text` and Discord\'s `content`; for Discord add `discord.com` to `egressAllow`; a\nGoogle Chat space webhook (`chat.googleapis.com`, also added to `egressAllow`) gets `{ text }` alone.\n\n## 9. Reading your other vxil projects\n\nAn integration of kind `vxil-project` reads another project with **that project\'s own key** \u2014 one key per\nproject, minted there with the least it needs (`usage:read` for its request meter, `cms:read` to aggregate its\ncollections). It is pull, not push: the other project does not know this dashboard exists, and revoking the key\nthere disconnects it. A second project gets a second secret: declare `secret:<name>` on `collect`, add the name\nto `PEER_SECRETS` in `functions/collect.ts`, and point the row at it with `config.peer` (any name not on that\nlist is refused, so a row cannot pick some other secret and send it as a bearer key).\n\n```json\n{ "key": "billing-project", "kind": "vxil-project", "enabled": true, "poll_every_min": 60,\n "config": { "aggregates": [ { "name": "signups_24h", "collection": "accounts", "fn": "count", "window_hours": 24 } ] } }\n```\n\n## 10. Limits and plans\n\n| | Free (staging / development projects) | Developer |\n|---|---|---|\n| Functions / crons | 5 functions, 3 crons, fastest every 15 min | 50 functions, 20 crons, every minute |\n| As shipped | set the two `*/5` schedules (`collect`, `evaluate-alerts`) to `*/15`; drop `drain-inbound`\'s cron trigger (its `payments.` trigger stays) | runs as shipped |\n| Requests a month | 100,000 | 1,000,000 |\n\nEvery call a function makes back into vxil counts as a request (vendor calls do not). The arithmetic for 3\npolled integrations \xD7 5 series, a 30-day month:\n\n| Function | Per run | Runs a month | Requests |\n|---|---|---|---|\n| `collect` every 5 min | 1 list + per integration (1 list + 5 tile patches + 1 status patch) = 22 | 8,640 | ~190,000 |\n| hour closes | 15 series \xD7 1 point | 720 hours | ~11,000 |\n| `evaluate-alerts` every 5 min | 1 rules read + 1 latest read (+ writes on crossings only) | 8,640 | ~17,000 |\n| `drain-inbound` every minute, 1 source | 1 list + 1 events read (+ 1 write per event) | 43,200 | ~86,000 + events |\n| `compact-daily` | 2 aggregates + 15 day rows + retention deletes | 30 | ~1,500 |\n\nAbout **305,000 requests a month** on Developer. On Free, at 15-minute cadence with 2 integrations \xD7 4 series and\nno inbound drain: `collect` ~37,000, `evaluate-alerts` ~6,000, hour closes ~6,000 \u2014 about **50,000**. A tile list\nread by your frontend counts too, so let realtime change-data tell the page when to re-read instead of polling.\n\n## 11. Compose with\n\n- **alerts-to-slack** \u2014 alerts on the platform\'s own failure events (a dead-lettered job, a quarantined\n function, a failing payments webhook) next to this dashboard\'s business thresholds.\n- **service-monitor** \u2014 uptime checks for your endpoints, written into the same kind of rows.\n- **team-workspace** \u2014 the full org model: invitations, per-resource grants, a field only finance receives.\n\n## 12. Where the frontend lives\n\nThe **App builder** builds public-facing apps. An internal dashboard is your own frontend \u2014 any framework,\nany host \u2014 reading this project with the two keys above. The reference is `examples/apps/team-dashboard`:\nsign-in through your identity provider, tiles that re-read on change-data frames, trend charts over\n`metric_points`, the feed, and the incident buttons calling `ops-action`.\n',
|
|
18471
18513
|
"functions": {
|
|
18472
18514
|
"collect.ts": "// collect.ts \u2014 PULL: poll the team's services into the tiles and the hourly trend\n// (a vxil function; cron every 5 minutes + an http \"Sync now\" door).\n//\n// There is no connectors feature. A connector is the `adapter` switch below +\n// a secret + a host in egressAllow. Each tick:\n// 1. reads the enabled integrations whose next_due_at has passed (oldest\n// first, at most MAX_PER_TICK);\n// 2. runs that integration's adapter \u2014 stripe, rest-json or vxil-project \u2014\n// which returns samples: { series, value, unit, sensitive };\n// 3. per series: when the hour of the stored sample has CLOSED, writes that\n// hour's closing sample as ONE metric_points row (grain 'hour'; the unique\n// point_key makes a redelivered tick a 409, never a duplicate) and records\n// it in last_bucket; then PATCHes metric_latest with the new sample (the\n// first sighting creates it);\n// 4. PATCHes the integration: ok + next_due_at, or broken + the error.\n//\n// A sensitive series (revenue, balances) writes `restricted_value` \u2014 readable\n// only by the roles in the config's readRoles \u2014 and leaves `value` null.\n//\n// A vendor failure never throws: the integration is marked broken (failures\n// counted, last_error kept to 500 chars) and the tick answers 200, so one\n// vendor's outage does not trip the function's circuit breaker.\n//\n// \"Sync now\" (http): `POST /v1/fn/collect { integration: '<key>' }` with a\n// signed-in staff session. The caller must hold `integrations.sync` in the ops\n// org (checked with orgs, not trusted from the browser). The http run updates\n// the tiles; a series whose hour has closed is left to the next cron tick,\n// which can read the restricted value it needs to close that hour.\n// A server caller (no session \u2014 `vxil functions invoke collect`, your backend)\n// runs one named integration, or the whole tick when none is named.\n//\n// cron-walk: drains-filter \u2014 each processed integration is PATCHed with next_due_at = now + poll_every_min, so it leaves the read filter\n\nimport type { CronFunctionEnvelope, HttpFunctionEnvelope } from '@vxil/sdk';\n\nconst MAX_PER_TICK = 20; // integrations per tick\nconst MAX_SERIES_PER_SOURCE = 25; // samples one adapter may emit\nconst ERROR_MAX = 500;\nconst OPS_ORG_SLUG = 'ops'; // the one org every staff member joins (README step 3)\nconst VENDOR_TIMEOUT_MS = 8_000;\n\n// \u2500\u2500 credentials are bound to hosts HERE, in code \u2014 never by a row \u2500\u2500\n// An integration row is data a dashboard admin can edit, and its base_url may\n// name any host in egressAllow. So a row never decides where a credential goes:\n// each secret is sent to the ONE host written below, and a row that points it\n// anywhere else turns `broken` without a request being made.\n// shop_token \u2192 the host of your REST API (edit to yours; the demo row uses \"auth\": \"none\")\n// stripe_key \u2192 api.stripe.com, hardcoded in its adapter\n// peer keys \u2192 this platform's own API only, and only the names listed in PEER_SECRETS\nconst SECRET_HOST: Record<string, string> = { shop_token: 'shop.example.com' };\n// The secrets a `vxil-project` row may name in config.peer. A second project =\n// a second `secret:<name>` on collect AND its name here.\nconst PEER_SECRETS: readonly string[] = ['vxil_peer_key'];\n\ntype Envelope = CronFunctionEnvelope | HttpFunctionEnvelope<{ integration?: string }>;\n\ninterface SeriesDef { name: string; json_path: string; unit?: string; sensitive?: boolean; label?: string }\ninterface AggregateDef { name: string; collection: string; fn: 'count' | 'sum' | 'avg' | 'min' | 'max'; field?: string; filter?: Record<string, unknown>; window_hours?: number; sensitive?: boolean; unit?: string }\ninterface IntegrationConfig {\n base_url?: string; path?: string; auth?: 'bearer' | 'none'; series?: SeriesDef[];\n aggregates?: AggregateDef[]; peer?: string;\n}\ninterface Integration {\n key: string; kind: string; status?: string; enabled?: boolean; poll_every_min?: number;\n consecutive_failures?: number; config?: IntegrationConfig; label?: string;\n}\ninterface Latest {\n series: string; source?: string; status?: string; last_bucket?: string | null; value?: number | null;\n restricted_value?: number | null; updated_at?: string; label?: string; unit?: string; sensitive?: boolean;\n hour_samples?: number; hour_min?: number | null; hour_max?: number | null;\n}\ninterface Item<T> { item_id: string; version?: number; data: T }\ninterface Sample { series: string; value: number; unit?: string | undefined; label?: string | undefined; sensitive?: boolean | undefined }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Envelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cmsJwt = env.scoped_jwts?.cms;\n if (!cmsJwt) return json({ error: 'missing cms scope' }, 403);\n const cms = new Cms(base, cmsJwt);\n const now = new Date();\n\n // \u2500\u2500 the http door \u2500\u2500\n // a signed-in staff member (\"Sync now\"): checked against the ops org, one integration\n // a server caller (`vxil functions invoke collect`, your own backend): one\n // integration when named, otherwise exactly what a cron tick does\n const key = env.trigger === 'http' && typeof env.payload?.integration === 'string' ? env.payload.integration : '';\n if (env.trigger === 'http' && (env.end_user || key)) {\n const user = env.end_user;\n if (user && !(await orgAllows(base, env.scoped_jwts?.orgs, user.id, 'integrations.sync'))) {\n return json({ error: 'role_required', permission: 'integrations.sync' }, 403);\n }\n const row = key ? (await cms.list<Integration>('integrations', { key }, { limit: 1 })).items[0] : undefined;\n if (!row) return json({ error: 'unknown_integration' }, 404);\n if (row.data.kind === 'webhook-source') return json({ error: 'webhook sources are drained, not polled' }, 422);\n const out = await pollOne(cms, env.secrets ?? {}, base, row, now, user ? 'http' : 'cron');\n return json(out, 200);\n }\n\n // \u2500\u2500 the cron tick \u2500\u2500\n const due = await cms.list<Integration>('integrations', {\n enabled: true,\n kind: { $in: ['stripe', 'rest-json', 'vxil-project'] },\n next_due_at: { $lte: now.toISOString() },\n }, { sort: 'next_due_at', limit: MAX_PER_TICK });\n const results = [];\n for (const row of due.items) results.push(await pollOne(cms, env.secrets ?? {}, base, row, now, 'cron'));\n return json({\n polled: results.length,\n ok: results.filter((r) => r.ok).length,\n broken: results.filter((r) => !r.ok).map((r) => r.integration),\n series: results.reduce((n, r) => n + r.series, 0),\n points: results.reduce((n, r) => n + r.points, 0),\n // more were due than one tick takes; the rest are first in line next tick\n truncated: due.items.length === MAX_PER_TICK,\n }, 200);\n },\n};\n\n/** One integration: adapter \u2192 samples \u2192 tiles + closed hours \u2192 the integration's own row. */\nasync function pollOne(\n cms: Cms, secrets: Record<string, string>, base: string, row: Item<Integration>, now: Date, mode: 'cron' | 'http',\n): Promise<{ integration: string; ok: boolean; series: number; points: number; deferred: number; error?: string }> {\n const integ = row.data;\n const nextDue = new Date(now.getTime() + Math.max(1, integ.poll_every_min ?? 5) * 60_000).toISOString();\n let samples: Sample[];\n try {\n samples = (await adapter(integ, secrets, base, now)).slice(0, MAX_SERIES_PER_SOURCE);\n } catch (e) {\n const missing = e instanceof MissingSecret;\n const error = String((e as Error).message ?? e).slice(0, ERROR_MAX);\n // If-Match: the row was read this tick, so the failure count stays exact\n // when two ticks overlap (the loser's PATCH is a 409 and changes nothing).\n await cms.patch('integrations', row.item_id, {\n status: missing ? 'could_not_check' : 'broken',\n consecutive_failures: (integ.consecutive_failures ?? 0) + 1,\n last_error: error,\n last_error_at: now.toISOString(),\n next_due_at: nextDue,\n }, row.version);\n return { integration: integ.key, ok: false, series: 0, points: 0, deferred: 0, error };\n }\n\n const w = await writeSamples(cms, integ.key, samples, now, mode);\n await cms.patch('integrations', row.item_id, {\n status: 'ok', consecutive_failures: 0, last_ok_at: now.toISOString(), last_error: null, next_due_at: nextDue,\n }, row.version);\n return { integration: integ.key, ok: true, series: w.series, points: w.points, deferred: w.deferred };\n}\n\n/** Write one source's samples: close finished hours, then update the tiles. */\nasync function writeSamples(\n cms: Cms, source: string, samples: Sample[], now: Date, mode: 'cron' | 'http',\n): Promise<{ series: number; points: number; deferred: number }> {\n const nowIso = now.toISOString();\n const hourNow = nowIso.slice(0, 13);\n const existing = new Map<string, Item<Latest>>();\n for (const it of (await cms.list<Latest>('metric_latest', { source }, { limit: 100 })).items) existing.set(it.data.series, it);\n\n let series = 0;\n let points = 0;\n let deferred = 0;\n for (const s of samples) {\n const cur = existing.get(s.series);\n const sensitive = Boolean(s.sensitive);\n const reading = sensitive ? { value: null, restricted_value: s.value } : { value: s.value, restricted_value: null };\n\n if (!cur) {\n // first sighting \u2014 a racing tick's create is a 409; the next tick patches it\n const r = await cms.create('metric_latest', {\n series: s.series, source, status: 'ok', label: s.label ?? s.series, unit: s.unit ?? null, sensitive,\n ...(sensitive ? { restricted_value: s.value } : { value: s.value, hour_min: s.value, hour_max: s.value }),\n hour_samples: 1, updated_at: nowIso,\n });\n if (r.ok) series++;\n continue;\n }\n\n const prevHour = (cur.data.updated_at ?? '').slice(0, 13);\n const hourClosed = prevHour !== '' && prevHour < hourNow;\n const patch: Record<string, unknown> = {\n ...reading, updated_at: nowIso, label: s.label ?? cur.data.label ?? s.series, unit: s.unit ?? cur.data.unit ?? null, sensitive,\n };\n if (cur.data.status === 'stale') patch.status = 'ok'; // fresh data again; alert severities are evaluate-alerts' to set\n\n if (hourClosed) {\n if (mode === 'http') { deferred++; continue; } // the cron closes this hour first\n if (cur.data.last_bucket !== prevHour) {\n // the closed hour's closing sample \u2192 ONE point (409 on point_key = already written)\n const wasSensitive = Boolean(cur.data.sensitive);\n const closing = wasSensitive ? cur.data.restricted_value : cur.data.value;\n if (typeof closing === 'number') {\n const p = await cms.create('metric_points', {\n series: s.series, grain: 'hour', ts: `${prevHour}:00:00.000Z`,\n // bucket + point_key are derived by the config's hooks; sent too so the\n // row is complete even where hooks are switched off\n bucket: prevHour, point_key: `${s.series}|hour|${prevHour}`,\n ...(wasSensitive\n ? { restricted_value: closing }\n : { value: closing, min: cur.data.hour_min ?? closing, max: cur.data.hour_max ?? closing }),\n samples: cur.data.hour_samples ?? 1, unit: cur.data.unit ?? null,\n });\n if (p.ok) points++;\n if (p.ok || p.status === 409) patch.last_bucket = prevHour;\n } else {\n patch.last_bucket = prevHour; // nothing to close (the stored value was gated or empty)\n }\n }\n patch.hour_samples = 1;\n patch.hour_min = sensitive ? null : s.value;\n patch.hour_max = sensitive ? null : s.value;\n } else {\n patch.hour_samples = (cur.data.hour_samples ?? 0) + 1;\n patch.hour_min = sensitive ? null : Math.min(cur.data.hour_min ?? s.value, s.value);\n patch.hour_max = sensitive ? null : Math.max(cur.data.hour_max ?? s.value, s.value);\n }\n const r = await cms.patch('metric_latest', cur.item_id, patch, cur.version);\n if (r.ok) series++;\n }\n return { series, points, deferred };\n}\n\n// \u2500\u2500 the adapters: one case per kind, self-contained (a blueprint is a clone) \u2500\u2500\n\nclass MissingSecret extends Error {}\n\nasync function adapter(integ: Integration, secrets: Record<string, string>, vxilBase: string, now: Date): Promise<Sample[]> {\n const cfg = integ.config ?? {};\n switch (integ.kind) {\n case 'stripe': {\n // A RESTRICTED key, read-only on Balance and Charges. Amounts are minor units.\n const key = secrets.stripe_key;\n if (!key) throw new MissingSecret('secret stripe_key is not set');\n const auth = { authorization: `Bearer ${key}` };\n const balance = await getJson(`https://api.stripe.com/v1/balance`, auth) as { available?: Array<{ amount: number; currency: string }> };\n const since = Math.floor(now.getTime() / 1000) - 3600;\n const charges = await getJson(`https://api.stripe.com/v1/charges?limit=100&created[gte]=${since}`, auth) as {\n data?: Array<{ amount: number; currency: string; status: string }>; has_more?: boolean;\n };\n const list = charges.data ?? [];\n const ok = list.filter((c) => c.status === 'succeeded');\n const failed = list.filter((c) => c.status === 'failed').length;\n const out: Sample[] = [\n { series: 'stripe:charges_1h', value: ok.length, unit: 'count', label: 'Charges (last hour)' },\n { series: 'stripe:failed_charges_1h', value: failed, unit: 'count', label: 'Failed charges (last hour)' },\n {\n series: 'stripe:charge_error_rate_1h', unit: '%', label: 'Charge error rate (last hour)',\n value: list.length === 0 ? 0 : Math.round((failed / list.length) * 1000) / 10,\n },\n ];\n const volume = new Map<string, number>();\n for (const c of ok) volume.set(c.currency, (volume.get(c.currency) ?? 0) + c.amount);\n for (const [cur, amount] of volume) {\n out.push({ series: `stripe:volume_1h:${cur}`, value: amount / 100, unit: cur, label: `Volume ${cur.toUpperCase()} (last hour)`, sensitive: true });\n }\n for (const b of balance.available ?? []) {\n out.push({ series: `stripe:balance:${b.currency}`, value: b.amount / 100, unit: b.currency, label: `Available ${b.currency.toUpperCase()}`, sensitive: true });\n }\n // more than 100 charges in the hour: the counts are a floor \u2014 page with\n // `starting_after` (and keep the cursor on the integration row) if you need exact ones\n return out;\n }\n\n case 'rest-json': {\n // Any JSON endpoint: GET base_url + path, pick numbers with json_path.\n if (!cfg.base_url || !cfg.series?.length) throw new Error('config needs base_url and series[]');\n let url: URL;\n try { url = new URL(`${cfg.base_url.replace(/\\/$/, '')}${cfg.path ?? ''}`); } catch { throw new Error('config base_url + path is not a URL'); }\n if (url.protocol !== 'https:') throw new Error('config base_url must be https://');\n const headers: Record<string, string> = { accept: 'application/json' };\n if (cfg.auth !== 'none') {\n // the host of the FINAL url (a path like \"@other.host\" cannot move it)\n if (url.host !== SECRET_HOST.shop_token) {\n throw new Error(`shop_token is bound to ${SECRET_HOST.shop_token} and is never sent to ${url.host}: set \"auth\": \"none\" for a public endpoint, or change SECRET_HOST in collect.ts`);\n }\n if (!secrets.shop_token) throw new MissingSecret('secret shop_token is not set');\n headers.authorization = `Bearer ${secrets.shop_token}`;\n }\n const body = await getJson(url.href, headers);\n const out: Sample[] = [];\n for (const s of cfg.series) {\n const v = pick(body, s.json_path);\n if (typeof v === 'number' && Number.isFinite(v)) {\n out.push({ series: `${integ.key}:${s.name}`, value: v, unit: s.unit, label: s.label ?? s.name, sensitive: s.sensitive });\n }\n }\n if (out.length === 0) throw new Error(`no numeric value at ${cfg.series.map((s) => s.json_path).join(', ')}`);\n return out;\n }\n\n case 'vxil-project': {\n // Another vxil project of yours, read with ITS key (one secret per project;\n // the least-privilege pattern \u2014 usage:read, plus cms:read for aggregates).\n const secretName = cfg.peer ?? 'vxil_peer_key';\n if (!PEER_SECRETS.includes(secretName)) {\n throw new Error(`config.peer '${secretName}' is not a peer secret: add it to PEER_SECRETS in collect.ts`);\n }\n const key = secrets[secretName];\n if (!key) throw new MissingSecret(`secret ${secretName} is not set`);\n const H = { authorization: `Bearer ${key}`, 'content-type': 'application/json' };\n const usage = await getJson(`${vxilBase}/v1/usage`, H) as { data?: { requests?: { used?: number } } };\n const out: Sample[] = [];\n if (typeof usage.data?.requests?.used === 'number') {\n out.push({ series: `${integ.key}:requests_mtd`, value: usage.data.requests.used, unit: 'requests', label: 'Requests this month' });\n }\n for (const a of cfg.aggregates ?? []) {\n const since = new Date(now.getTime() - (a.window_hours ?? 24) * 3600_000).toISOString();\n const r = await fetch(`${vxilBase}/v1/cms/items/${encodeURIComponent(a.collection)}/aggregate`, {\n method: 'POST', headers: H, signal: AbortSignal.timeout(VENDOR_TIMEOUT_MS),\n body: JSON.stringify({\n aggregates: [a.fn === 'count' ? { fn: 'count', as: 'v' } : { fn: a.fn, field: a.field, as: 'v' }],\n ...(a.filter ? { filter: a.filter } : {}),\n window: { field: 'created_at', since },\n }),\n });\n if (!r.ok) throw new Error(`aggregate ${a.collection}: ${r.status}`);\n const g = ((await r.json()) as { data?: { groups?: Array<{ v?: number }> } }).data?.groups?.[0];\n out.push({ series: `${integ.key}:${a.name}`, value: Number(g?.v ?? 0), unit: a.unit, label: a.name, sensitive: a.sensitive });\n }\n return out;\n }\n\n default:\n throw new Error(`no adapter for kind '${integ.kind}'`);\n }\n}\n\nasync function getJson(url: string, headers: Record<string, string>): Promise<unknown> {\n const r = await fetch(url, { headers, signal: AbortSignal.timeout(VENDOR_TIMEOUT_MS) });\n if (r.status === 403 && r.headers.get('x-vxil-egress') === 'blocked') {\n throw new Error(`egress blocked: add ${new URL(url).hostname} to egressAllow and push`);\n }\n if (!r.ok) throw new Error(`GET ${new URL(url).hostname}${new URL(url).pathname}: ${r.status}`);\n return r.json();\n}\n\n/** 'a.b.0.c' into a JSON value; a trailing '.length' counts an array. */\nfunction pick(root: unknown, path: string): unknown {\n let cur: unknown = root;\n for (const part of path.split('.')) {\n if (part === 'length' && Array.isArray(cur)) return cur.length;\n if (cur === null || typeof cur !== 'object') return undefined;\n cur = (cur as Record<string, unknown>)[part];\n }\n return cur;\n}\n\n/** Revocation-grade check: is this user allowed `permission` in the ops org? */\nasync function orgAllows(base: string, orgsJwt: string | undefined, userId: string, permission: string): Promise<boolean> {\n if (!orgsJwt) return false;\n const H = { authorization: `Bearer ${orgsJwt}` };\n const mine = await fetch(`${base}/v1/orgs?user_id=${encodeURIComponent(userId)}`, { headers: H });\n if (!mine.ok) return false;\n const org = ((await mine.json()) as { data?: { orgs?: Array<{ org_id: string; slug: string }> } }).data?.orgs\n ?.find((o) => o.slug === OPS_ORG_SLUG);\n if (!org) return false;\n const q = `user_id=${encodeURIComponent(userId)}&permission=${encodeURIComponent(permission)}`;\n const check = await fetch(`${base}/v1/orgs/${encodeURIComponent(org.org_id)}/check?${q}`, { headers: H });\n if (!check.ok) return false;\n return ((await check.json()) as { data?: { allowed?: boolean } }).data?.allowed === true;\n}\n\n/** The cms REST envelope: reads { data: { items: [{ item_id, version, data }], next_cursor } }. */\nclass Cms {\n constructor(private base: string, private jwt: string) {}\n private h(extra: Record<string, string> = {}) {\n return { authorization: `Bearer ${this.jwt}`, 'content-type': 'application/json', ...extra };\n }\n async list<T>(coll: string, filter: Record<string, unknown>, o: { sort?: string; limit?: number } = {}): Promise<{ items: Item<T>[]; next_cursor: string | null }> {\n const q = new URLSearchParams({ filter: JSON.stringify(filter), limit: String(o.limit ?? 100) });\n if (o.sort) q.set('sort', o.sort);\n const r = await fetch(`${this.base}/v1/cms/items/${coll}?${q}`, { headers: this.h() });\n if (!r.ok) throw new Error(`cms list ${coll}: ${r.status}`);\n const d = ((await r.json()) as { data?: { items?: Item<T>[]; next_cursor?: string | null } }).data;\n return { items: d?.items ?? [], next_cursor: d?.next_cursor ?? null };\n }\n async create(coll: string, data: Record<string, unknown>): Promise<{ ok: boolean; status: number }> {\n const r = await fetch(`${this.base}/v1/cms/items/${coll}`, {\n method: 'POST', headers: this.h(), body: JSON.stringify({ status: 'published', data }),\n });\n return { ok: r.ok, status: r.status };\n }\n async patch(coll: string, id: string, data: Record<string, unknown>, version?: number): Promise<{ ok: boolean; status: number }> {\n const r = await fetch(`${this.base}/v1/cms/items/${coll}/${id}`, {\n method: 'PATCH',\n headers: this.h(version !== undefined ? { 'if-match': String(version) } : {}),\n body: JSON.stringify({ data }),\n });\n return { ok: r.ok, status: r.status };\n }\n}\n\nconst json = (o: unknown, status: number) => Response.json(o, { status });\n",
|
|
18473
18515
|
"compact-daily.ts": "// compact-daily.ts \u2014 hour rows \u2192 day rows, then retention (a vxil function;\n// cron '15 0 * * *', just after midnight UTC).\n//\n// 1. Two bounded server-side aggregates over YESTERDAY's hour rows\n// (POST /v1/cms/items/metric_points/aggregate, grouped by series, windowed\n// on the `ts` slot): avg/min/max/count of `value`, and avg/max of the gated\n// `restricted_value` (this function runs as the server, so it sees both).\n// 2. One grain 'day' row per series. The unique point_key ('series|day|date')\n// makes a re-run a 409 per row \u2014 the day is never written twice.\n// 3. Retention: hour rows older than RETAIN_HOUR_DAYS and feed rows older than\n// RETAIN_FEED_DAYS are removed with the bounded filtered delete (100 rows a\n// call, following next_cursor, at most MAX_DELETE_CALLS calls a night; the\n// remainder is reported as `truncated` and taken the next night).\n// 4. The `__compact__` row in ops_state records the last day finished.\n//\n// Why hour buckets and not raw samples: every cms write is an audited row and\n// a change-data frame. One closed hour per series is 24 rows a day; a raw\n// sample every 5 minutes would be 288 \u2014 twelve times the writes, the frames,\n// the request budget, and the rows the 50,000-row aggregate scan must cover.\n//\n// cron-walk: complete-per-tick \u2014 one aggregate per value kind is the whole read; the retention deletes follow next_cursor under MAX_DELETE_CALLS and report `truncated`\n\nimport type { CronFunctionEnvelope } from '@vxil/sdk';\n\nconst RETAIN_HOUR_DAYS = 90;\nconst RETAIN_FEED_DAYS = 30;\nconst MAX_DELETE_CALLS = 30;\nconst MAX_GROUPS = 500;\n\ninterface Group { key: { series?: string }; avg?: number | null; min?: number | null; max?: number | null; n?: number }\ninterface Item<T> { item_id: string; version?: number; data: T }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as CronFunctionEnvelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cmsJwt = env.scoped_jwts?.cms;\n if (!cmsJwt) return Response.json({ error: 'missing cms scope' }, { status: 403 });\n const H = { authorization: `Bearer ${cmsJwt}`, 'content-type': 'application/json' };\n\n // the slot this tick was due for (stable across a late start or a re-delivery)\n const slot = new Date(env.scheduled_for ?? Date.now());\n const day = new Date(Date.UTC(slot.getUTCFullYear(), slot.getUTCMonth(), slot.getUTCDate() - 1)).toISOString().slice(0, 10);\n const since = `${day}T00:00:00.000Z`;\n const until = new Date(Date.parse(since) + 86_400_000).toISOString(); // exclusive\n\n const aggregate = async (field: 'value' | 'restricted_value'): Promise<Group[] | null> => {\n const r = await fetch(`${base}/v1/cms/items/metric_points/aggregate`, {\n method: 'POST', headers: H,\n body: JSON.stringify({\n aggregates: [\n { fn: 'avg', field, as: 'avg' }, { fn: 'min', field, as: 'min' },\n { fn: 'max', field, as: 'max' }, { fn: 'count', as: 'n' },\n ],\n groupBy: ['series'],\n filter: { grain: 'hour' },\n window: { field: 'ts', since, until },\n limit: MAX_GROUPS,\n }),\n });\n if (!r.ok) return null;\n return ((await r.json()) as { data?: { groups?: Group[] } }).data?.groups ?? [];\n };\n const plain = await aggregate('value');\n const gated = await aggregate('restricted_value');\n if (!plain || !gated) return Response.json({ error: 'aggregate_failed', day }, { status: 502 });\n\n // one day row per series; a series whose hours carried restricted_value is sensitive\n const bySeries = new Map<string, { plain?: Group; gated?: Group }>();\n for (const g of plain) if (g.key.series) bySeries.set(g.key.series, { plain: g });\n for (const g of gated) if (g.key.series) bySeries.set(g.key.series, { ...bySeries.get(g.key.series), gated: g });\n\n let written = 0;\n let existed = 0;\n for (const [series, { plain: p, gated: q }] of bySeries) {\n const sensitive = typeof q?.avg === 'number';\n if (!sensitive && typeof p?.avg !== 'number') continue;\n const r = await fetch(`${base}/v1/cms/items/metric_points`, {\n method: 'POST', headers: H,\n body: JSON.stringify({\n status: 'published',\n data: {\n series, grain: 'day', ts: since, bucket: day, point_key: `${series}|day|${day}`,\n samples: p?.n ?? q?.n ?? 0,\n ...(sensitive\n ? { restricted_value: round(q!.avg!) }\n : { value: round(p!.avg!), min: p!.min ?? null, max: p!.max ?? null }),\n },\n }),\n });\n if (r.ok) written++;\n else if (r.status === 409) existed++;\n }\n\n // retention \u2014 the bounded filtered delete, followed to the end or the cap\n const cutoff = (days: number) => new Date(Date.parse(since) - days * 86_400_000).toISOString();\n let calls = 0;\n const sweep = async (coll: string, filter: Record<string, unknown>): Promise<{ deleted: number; complete: boolean }> => {\n let deleted = 0;\n let cursor: string | null = null;\n while (calls < MAX_DELETE_CALLS) {\n calls++;\n const r = await fetch(`${base}/v1/cms/items/${coll}/delete`, {\n method: 'POST', headers: H,\n body: JSON.stringify({ filter, limit: 100, ...(cursor ? { cursor } : {}) }),\n });\n if (!r.ok) return { deleted, complete: false };\n const d = ((await r.json()) as { data?: { deleted?: number; complete?: boolean; next_cursor?: string | null } }).data ?? {};\n deleted += d.deleted ?? 0;\n // a call stopped by its time budget hands back next_cursor; otherwise the\n // filter simply re-matches what is left \u2014 loop while rows are still going\n cursor = d.complete === false ? (d.next_cursor ?? null) : null;\n if ((d.deleted ?? 0) === 0) return { deleted, complete: true };\n }\n return { deleted, complete: false };\n };\n const hours = await sweep('metric_points', { grain: 'hour', ts: { $lt: cutoff(RETAIN_HOUR_DAYS) } });\n const feed = await sweep('events_feed', { occurred_at: { $lt: cutoff(RETAIN_FEED_DAYS) } });\n\n // remember the finished day (create once, then PATCH)\n const sq = new URLSearchParams({ filter: JSON.stringify({ key: '__compact__' }), limit: '1' });\n const st = await fetch(`${base}/v1/cms/items/ops_state?${sq}`, { headers: H });\n const row = ((await st.json().catch(() => ({}))) as { data?: { items?: Item<{ cursor?: string }>[] } }).data?.items?.[0];\n const mark = { key: '__compact__', cursor: day, updated_at: new Date().toISOString() };\n if (row) {\n await fetch(`${base}/v1/cms/items/ops_state/${row.item_id}`, { method: 'PATCH', headers: H, body: JSON.stringify({ data: mark }) });\n } else {\n await fetch(`${base}/v1/cms/items/ops_state`, { method: 'POST', headers: H, body: JSON.stringify({ status: 'published', data: mark }) });\n }\n\n return Response.json({\n day, series: bySeries.size, written, existed,\n groups_capped: plain.length >= MAX_GROUPS || gated.length >= MAX_GROUPS,\n pruned: { hour_rows: hours.deleted, feed_rows: feed.deleted },\n truncated: !hours.complete || !feed.complete, // retention resumes tomorrow\n });\n },\n};\n\nconst round = (n: number) => Math.round(n * 1000) / 1000;\n",
|
|
@@ -18503,7 +18545,7 @@ export default defineConfig({
|
|
|
18503
18545
|
"oidc_client_secret"
|
|
18504
18546
|
],
|
|
18505
18547
|
"configSrc": "import { defineConfig } from '@vxil/config';\n\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n// \"Service monitor\" \u2014 uptime and health checks for your team's PUBLIC\n// endpoints, dead-man's-switch heartbeats from your own jobs, outages that open\n// and close once per state crossing, the on-call shift paged from data, and a\n// keyless public status page. Declared end-to-end in ONE typed file.\n//\n// \u2022 auth \u2192 staff sign in through your identity provider (OIDC);\n// password and magic-link sign-in are off \u2014 staff only\n// \u2022 orgs \u2192 one team org; the roles `oncall` (outages.manage) and\n// `monitor-admin` (checks.manage) are rows you add once\n// \u2022 cms \u2192 checks \xB7 uptime_hourly \xB7 outages \xB7 oncall_shifts (private)\n// and status_components \xB7 outage_updates (\u2B22 PUBLIC: only\n// their PUBLISHED rows are readable with no key)\n// \u2022 notifications \u2192 the page: e-mail + the in-app inbox of the shift\n// \u2022 functions \u2192 probe (cron) \xB7 heartbeat (http) \xB7 outage-action (http) \xB7\n// daily-report (cron)\n//\n// There is no monitoring engine, no escalation engine and no connector here:\n// a probe is a cron function behind a deny-by-default egress allowlist, an\n// outage is a row whose creation is guarded, a page is a read of who is on\n// shift right now, and escalation is one more filter in the same cron.\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\nexport default defineConfig({\n env: 'staging',\n\n features: {\n // \u2500\u2500 Staff-only sign-in \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n // The OIDC issuer is your company's identity provider; `allowedDomains`\n // fences sign-in to your own e-mail domain (no e-mail \u21D2 refused). Password\n // and magic-link sign-in are switched off, so nobody outside the IdP can\n // create an account. Store the client secret once:\n // printf '%s' \"$OIDC_SECRET\" | vxil secrets set auth/oidc_client_secret\n auth: {\n methods: { emailPassword: false, magicLink: false },\n providers: {\n oidc: {\n issuer: 'https://login.example-idp.com', // https, no query/fragment\n clientId: 'vxil-service-monitor', // not a secret: it rides every authorize URL\n clientSecretRef: 'secret:oidc_client_secret', // the `secrets` block below\n scopes: ['email', 'profile'], // `openid` is always added\n allowedDomains: ['example.com'], // fail-closed domain fence\n },\n },\n // The member's org role rides the session (a snapshot, refreshed with it).\n // The functions below do NOT trust it: they re-check the permission live.\n orgClaims: { enabled: true },\n session: { ttlMinutes: 60, refreshTtlDays: 7 },\n security: {\n // every sign-in return URL must be your own admin app\n allowedRedirectOrigins: ['https://status-admin.example.com'],\n },\n },\n\n // \u2500\u2500 One team org, three roles \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n // `viewer` is built in (reads only). `oncall` and `monitor-admin` are\n // CUSTOM roles \u2014 rows, not config \u2014 added once with `POST /v1/orgs/roles`\n // (see the README). `owner` holds every permission.\n orgs: { enabled: true, maxOrgsPerTenant: 5, maxMembersPerOrg: 500, invitationTtlHours: 72 },\n\n cms: {\n // Draft \u2192 publish is ON, and it matters for exactly one collection:\n // `outage_updates`. A staff note is written as a DRAFT and is never\n // public; only an update the team chooses to publish reaches the status\n // page. Every other write in this blueprint passes `status` explicitly.\n draftPublish: true,\n hooks: {\n // The OUTAGE LIFECYCLE, enforced in the same write: open \u2192 acked \u2192\n // resolved, open \u2192 resolved; `resolved` is terminal.\n outage_stage: {\n collection: 'outages',\n event: 'beforeUpdate',\n kind: 'validate',\n expr:\n 'item.state == before.state'\n + \" || (before.state == 'open' && (item.state == 'acked' || item.state == 'resolved'))\"\n + \" || (before.state == 'acked' && item.state == 'resolved')\",\n message: 'illegal outage state transition (open \u2192 acked \u2192 resolved)',\n },\n // ONE uptime row per check per hour: the composed key is DERIVED\n // server-side, and `point_key` is unique \u2014 a racing second create is\n // a 409, never a second row. (Hooks do not run on $inc, which is fine:\n // an increment never changes check or bucket.)\n uptime_key: {\n collection: 'uptime_hourly',\n event: 'beforeWrite',\n kind: 'derive',\n field: 'point_key',\n expr: \"concat(item.check, '|', item.bucket)\",\n },\n check_kind: {\n collection: 'checks',\n event: 'beforeWrite',\n kind: 'validate',\n expr: \"item.kind == 'http' || item.kind == 'keyword' || item.kind == 'heartbeat'\",\n message: 'kind must be http, keyword or heartbeat',\n },\n // A probed URL is https, always. A heartbeat check has no URL.\n check_url: {\n collection: 'checks',\n event: 'beforeWrite',\n kind: 'validate',\n expr: \"item.kind == 'heartbeat' || startsWith(item.url, 'https://')\",\n message: 'an http or keyword check needs an https:// url',\n },\n },\n // OPTIONAL live view for your admin screen: every outage write is pushed to\n // a realtime channel (enable the realtime feature too). Safe with\n // `payload: 'full'` because no `outages` field declares readRoles.\n // cdc: {\n // outages_live: { collection: 'outages', channel: 'monitor:outages', payload: 'full' },\n // },\n },\n\n // The page goes out as the shipped `transactional` message (a subject + a\n // paragraph + a button to the outage) on e-mail AND the in-app inbox.\n notifications: {\n provider: 'mock', // switch to your e-mail provider before production\n fromEmail: 'monitor@example.com',\n fromName: 'Service monitor',\n inboxEnabled: true,\n },\n\n functions: { enabled: true },\n },\n\n // \u2500\u2500 Schema-as-code (\u22648 index slots per collection: s1\u2013s4/n1\u2013n2/t1\u2013t2) \u2500\u2500\u2500\u2500\u2500\u2500\n cms: {\n collections: {\n // What to watch. One row per endpoint or heartbeat.\n checks: {\n singular: 'check',\n fields: {\n key: { type: 'string', required: true, indexSlot: 's1', unique: true }, // 'api-health'\n kind: { type: 'string', required: true, indexSlot: 's2' }, // http | keyword | heartbeat (hook-checked)\n component: { type: 'string', indexSlot: 's3' }, // a status_components key\n state: { type: 'string', indexSlot: 's4', validation: { enum: ['up', 'degraded', 'down', 'paused'] } },\n consecutive_failures: { type: 'int', indexSlot: 'n1', validation: { min: 0 } },\n last_latency_ms: { type: 'float', indexSlot: 'n2' },\n next_due_at: { type: 'datetime', indexSlot: 't1' }, // the probe's work-queue filter\n last_checked_at: { type: 'datetime', indexSlot: 't2' },\n // \u2500\u2500 unslotted \u2500\u2500\n name: { type: 'string', validation: { max: 120 } },\n url: { type: 'string', validation: { max: 2000 } }, // https only (check_url hook)\n expect_status: { type: 'int', validation: { min: 100, max: 599 } }, // default 200\n expect_contains: { type: 'string', validation: { max: 200 } }, // keyword checks\n send_probe_token: { type: 'bool' }, // send `x-probe-token: <probe_token secret>`\n timeout_ms: { type: 'int', validation: { min: 1000, max: 15000 } }, // default 10000\n degraded_ms: { type: 'int', validation: { min: 1 } }, // slower than this \u21D2 degraded\n interval_min: { type: 'int', validation: { min: 1, max: 1440 } }, // default 1\n failures_to_down: { type: 'int', validation: { min: 1, max: 10 } }, // default 2\n enabled: { type: 'bool', indexed: true },\n heartbeat_grace_min: { type: 'int', validation: { min: 1, max: 10080 } }, // heartbeat checks\n last_heartbeat_at: { type: 'datetime' },\n hour_bucket: { type: 'string' }, // 'YYYY-MM-DDTHH' of hour_row\n hour_row: { type: 'string' }, // the current uptime_hourly item id\n hour_max: { type: 'float' }, // the slowest sample this hour (drives latency_max)\n open_outage: { type: 'string' }, // the outage this down-crossing opened\n last_error: { type: 'text' },\n },\n },\n\n // Hourly counters: ok / total per check per hour, incremented atomically.\n uptime_hourly: {\n singular: 'uptime_point',\n fields: {\n check: { type: 'string', required: true, indexSlot: 's1' }, // a checks key\n bucket: { type: 'string', required: true, indexSlot: 's2' }, // 'YYYY-MM-DDTHH' (UTC)\n point_key: { type: 'string', indexSlot: 's3', unique: true }, // derived: check|bucket\n ok: { type: 'int', indexSlot: 'n1', validation: { min: 0 } },\n total: { type: 'int', indexSlot: 'n2', validation: { min: 0 } },\n ts: { type: 'datetime', indexSlot: 't1' }, // the hour's start \u2014 the report window field\n latency_sum: { type: 'float' },\n latency_max: { type: 'float' },\n },\n },\n\n // One row per down-crossing. Created under a lock + guard, so a check can\n // never have two open outages, however often a tick is re-delivered.\n outages: {\n singular: 'outage',\n fields: {\n check: { type: 'string', required: true, indexSlot: 's1' },\n state: { type: 'string', indexSlot: 's2', validation: { enum: ['open', 'acked', 'resolved'] } },\n severity: { type: 'string', indexSlot: 's3', validation: { enum: ['minor', 'major'] } },\n acked_by: { type: 'string', indexSlot: 's4' }, // the acknowledging staff member\n opened_at: { type: 'datetime', indexSlot: 't1' },\n resolved_at: { type: 'datetime', indexSlot: 't2' },\n title: { type: 'string', validation: { max: 200 } },\n cause: { type: 'text' },\n duration_min: { type: 'float' },\n paged: { type: 'json' }, // who was paged, when \u2014 written once (null \u21D2 not yet)\n escalated_at: { type: 'datetime' }, // the secondary was paged\n },\n },\n\n // \u2B22 PUBLIC: the updates the team PUBLISHES. A note stays a draft \u2014 and a\n // draft is never served on the public lane.\n outage_updates: {\n singular: 'outage_update',\n public: true,\n fields: {\n outage: { type: 'relation', relationTo: 'outages', required: true, indexSlot: 's1' },\n kind: {\n type: 'string', required: true, indexSlot: 's2',\n validation: { enum: ['investigating', 'identified', 'monitoring', 'resolved', 'note'] },\n },\n // the author's user id: visible to signed-in staff with a role, never on\n // the public lane (a readRoles field is never served anonymously)\n author: { type: 'string', indexSlot: 's3', readRoles: ['owner', 'admin', 'viewer', 'oncall', 'monitor-admin'] },\n at: { type: 'datetime', indexSlot: 't1' },\n component: { type: 'string' }, // the status_components key it is about\n body: { type: 'text', validation: { max: 2000 } },\n },\n },\n\n // \u2B22 PUBLIC: the components on your status page and their current status.\n status_components: {\n singular: 'status_component',\n public: true,\n fields: {\n key: { type: 'string', required: true, indexSlot: 's1', unique: true }, // 'api'\n name: { type: 'string', required: true, indexSlot: 's2' }, // 'API'\n group: { type: 'string', indexSlot: 's3' },\n status: {\n type: 'string', indexSlot: 's4',\n validation: { enum: ['operational', 'degraded', 'partial_outage', 'major_outage'] },\n },\n sort: { type: 'int', indexSlot: 'n1' },\n updated_at: { type: 'datetime', indexSlot: 't1' },\n description: { type: 'text' },\n },\n },\n\n // Who is on call, as DATA. The probe pages whoever's shift covers now.\n oncall_shifts: {\n singular: 'oncall_shift',\n fields: {\n user: { type: 'string', required: true, indexSlot: 's1' }, // the staff member's user id\n email: { type: 'string', indexSlot: 's2' }, // optional delivery override\n role: { type: 'string', indexSlot: 's3', validation: { enum: ['primary', 'secondary'] } },\n starts_at: { type: 'datetime', required: true, indexSlot: 't1' },\n ends_at: { type: 'datetime', required: true, indexSlot: 't2' },\n },\n },\n },\n },\n\n functions: {\n // THE PROBE. Every minute: due checks \u2192 probe in parallel \u2192 counters \u2192\n // once-per-crossing outages, component status and pages \u2192 escalation.\n probe: {\n entry: './functions/probe.ts',\n // Free plan: '*/15 * * * *' (its fastest function cron). `overlap: 'skip'`:\n // a tick still running never runs beside the next one.\n trigger: { kind: 'cron', schedule: '* * * * *', overlap: 'skip' },\n scopes: ['cms:read', 'cms:write', 'notifications:send'],\n secrets: ['secret:slack_webhook_url', 'secret:probe_token'],\n // Deny-by-default egress: EVERY probed host must be listed here, plus the\n // chat host ('discord.com' for Discord, 'chat.googleapis.com' for a Google\n // Chat space webhook). Private, loopback and internal hosts are unreachable\n // by design \u2014 expose a public health endpoint guarded by the probe token.\n egressAllow: ['status.example.com', 'api.example.com', 'hooks.slack.com'],\n limits: { timeoutMs: 15000 }, // per outbound fetch; a check's own timeout_ms is \u2264 this\n },\n\n // DEAD-MAN'S SWITCH. Your own cron jobs call it when they finish:\n // POST /v1/fn/heartbeat {\"check\":\"nightly-backup\"}\n // with a key holding ONLY functions:invoke, kept in their CI secret store.\n heartbeat: {\n entry: './functions/heartbeat.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write'],\n egressAllow: [],\n },\n\n // STAFF ACTIONS: ack \xB7 resolve \xB7 post_update (outages.manage) and pause \xB7\n // resume \xB7 publish_component (checks.manage), each re-checked live against\n // the caller's org role.\n 'outage-action': {\n entry: './functions/outage-action.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write', 'orgs:read'],\n egressAllow: [],\n },\n\n // THE MORNING E-MAIL. 08:00 UTC \u2014 function crons are UTC; if you need local\n // time, enqueue it from a jobs schedule that carries a timezone instead.\n 'daily-report': {\n entry: './functions/daily-report.ts',\n trigger: { kind: 'cron', schedule: '0 8 * * *', overlap: 'skip' },\n scopes: ['cms:read', 'notifications:send'],\n egressAllow: [],\n },\n },\n\n // References only \u2014 values are stored once and never appear in this file.\n secrets: {\n slack_webhook_url: { feature: 'functions', description: 'Slack incoming webhook (or Discord / Google Chat space webhook) URL for pages' },\n probe_token: { feature: 'functions', description: 'shared token your public health endpoints require in x-probe-token' },\n oidc_client_secret: { feature: 'auth', description: 'OIDC client secret for your identity provider' },\n },\n\n // Seeds land as DRAFTS (a seed is a plain create). Publish the two components\n // once \u2014 `outage-action` `publish_component`, or the dashboard \u2014 and they\n // appear on the status page. The http checks land PAUSED (state 'paused',\n // enabled false) until their hosts are in the probe's egressAllow; turn each\n // on with `outage-action` `resume`, whose precondition is `state: 'paused'`.\n seed: {\n cms: [\n {\n collection: 'status_components',\n items: [\n { key: 'api', name: 'API', group: 'Core', status: 'operational', sort: 1, updated_at: '2026-01-01T00:00:00Z' },\n { key: 'website', name: 'Website', group: 'Core', status: 'operational', sort: 2, updated_at: '2026-01-01T00:00:00Z' },\n ],\n },\n {\n collection: 'checks',\n items: [\n {\n key: 'api-health', name: 'API health', kind: 'http', component: 'api', state: 'paused',\n url: 'https://api.example.com/health', expect_status: 200, send_probe_token: true,\n timeout_ms: 10000, degraded_ms: 2000, interval_min: 1, failures_to_down: 2,\n enabled: false, consecutive_failures: 0, next_due_at: '2026-01-01T00:00:00Z',\n },\n {\n key: 'website-home', name: 'Website', kind: 'keyword', component: 'website', state: 'paused',\n url: 'https://status.example.com/', expect_status: 200, expect_contains: '<title>',\n timeout_ms: 10000, degraded_ms: 3000, interval_min: 5, failures_to_down: 2,\n enabled: false, consecutive_failures: 0, next_due_at: '2026-01-01T00:00:00Z',\n },\n {\n key: 'nightly-backup', name: 'Nightly backup', kind: 'heartbeat', component: 'api', state: 'up',\n heartbeat_grace_min: 1500, interval_min: 15, failures_to_down: 1,\n enabled: true, consecutive_failures: 0, next_due_at: '2026-01-01T00:00:00Z',\n },\n ],\n },\n ],\n },\n});\n",
|
|
18506
|
-
"readme": '# Service monitor (ops)\n\nUptime and health checks for your team\'s **public endpoints**, **dead-man\'s-switch heartbeats** from your\nown scheduled jobs, outages that open and close **once per state crossing**, the on-call shift **paged from\ndata** in chat, e-mail and the in-app inbox, staff who acknowledge, resolve and post updates, and a\n**keyless public status page** that shows component status and only the updates you chose to publish.\n\nIt is four functions and six collections. There is no monitoring engine, no escalation engine and no\nconnector in it: a probe is a cron function behind a deny-by-default egress allowlist, an outage is a row\nwhose creation is guarded, a page is a read of who is on shift right now, and escalation is one more\nfilter in the same cron.\n\n```bash\nvxil init --template service-monitor\nvxil quickstart --env staging --no-push --invite <code> # or `vxil link <slug> --env staging`\nvxil projects workload <slug> staging # Free plan: functions deploy on staging/development\nprintf \'%s\' "$SLACK_URL" | vxil secrets set functions/slack_webhook_url\nprintf \'%s\' "$PROBE_TOKEN" | vxil secrets set functions/probe_token\nprintf \'%s\' "$OIDC_SECRET" | vxil secrets set auth/oidc_client_secret\nvxil push # collections, hooks, auth, orgs, the four functions\nvxil seed # 2 components, 2 http checks (paused), 1 heartbeat\n```\n\nThen, once:\n\n```bash\n# the two custom roles (rows, not config \u2014 `viewer` and `owner` are built in)\nvxil api POST /v1/orgs/roles --data \'{"role_key":"oncall","name":"On call","permissions":["outages.manage"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"monitor-admin","name":"Monitor admin","permissions":["outages.manage","checks.manage"]}\'\n# the team org (its owner holds every permission), then each staff member with a role\n# \u2014 a member\'s user id exists after their first sign-in\nvxil api POST /v1/orgs --data \'{"slug":"platform","name":"Platform team","owner_user_id":"<your user id>"}\'\nvxil api POST /v1/orgs/<org_id>/members --data \'{"user_id":"<user id>","role":"oncall"}\'\n# who is on call, as data\nvxil api POST /v1/cms/items/oncall_shifts --data \'{"status":"published","data":{"user":"<user id>","role":"primary","starts_at":"2026-10-06T08:00:00Z","ends_at":"2026-10-13T08:00:00Z"}}\'\n```\n\nEdit `issuer`, `clientId` and `allowedDomains` in `vxil.config.ts` to your identity provider before you push.\n\n**Both probe secrets must be set before the probe can run**: a function whose declared secret has no value\nis refused at invoke (`409 secret_missing`), so the cron would fail every tick. No chat? Delete\n`\'secret:slack_webhook_url\'` from the probe\'s `secrets` (the probe then pages by e-mail and inbox only). No\ntoken-guarded endpoint yet? Set `probe_token` to any random value, or delete it the same way.\n\n> **Plan note.** Free deploys functions to `staging` and `development` projects, caps a project at 5\n> functions and 3 crons, and runs a cron every 15 minutes at the fastest. This blueprint uses 4 functions\n> and 2 crons, so it fits \u2014 change the probe\'s `schedule` to `*/15 * * * *` on Free (the config comment\n> says so), or the deploy answers `402 plan_limit`. Developer and above run it every minute.\n\n## The collections\n\n| Collection | What it holds | Who reads it |\n|---|---|---|\n| `checks` | what to watch: `key`, `kind` (`http` \xB7 `keyword` \xB7 `heartbeat`), `url`, expected status/text, timeout, `degraded_ms`, `interval_min`, `failures_to_down`, and the live state (`up` \xB7 `degraded` \xB7 `down` \xB7 `paused`) | staff |\n| `uptime_hourly` | one row per check per hour: `ok`, `total`, `latency_sum`, `latency_max` | staff, the daily report |\n| `outages` | one row per down-crossing: `open` \u2192 `acked` \u2192 `resolved`, who acked, duration, who was paged | staff |\n| `outage_updates` \u2B22 | the updates staff write; **only published ones are public** | everyone (published only) |\n| `status_components` \u2B22 | the components on your status page and their status | everyone (published only) |\n| `oncall_shifts` | who is on call, `primary` or `secondary`, from when to when | staff, the probe |\n\nThree Lane-A hooks guard the data in the same write: `outage_stage` (open \u2192 acked \u2192 resolved, resolved is\nfinal), `check_kind` and `check_url` (an http or keyword check needs an `https://` url), and `uptime_key`,\nwhich derives `point_key = check|bucket` server-side on a field declared `unique` \u2014 so one hour can never\nhave two rows for the same check.\n\n## Probes: a cron function with deny-by-default egress\n\n`probe` runs every minute. It reads the due checks \u2014 `enabled`, `next_due_at \u2264 now`, oldest first, at most\n`MAX_PER_TICK` (40) \u2014 and probes them in parallel:\n\n- **http** \u2014 `GET url`, with the check\'s `timeout_ms` (\u2264 15 s), `redirect: \'manual\'`, and the\n `x-probe-token` header when the check sets `send_probe_token`. Up when the status equals `expect_status`\n (200 by default).\n- **keyword** \u2014 the same, and the body must contain `expect_contains`.\n- **heartbeat** \u2014 no fetch at all (below).\n\nA **redirect is a failure** with reason `redirect`. The probe never follows one: a public host can redirect\nto a private one, so the platform\'s egress guard refuses to follow redirects, and the probe reports it\nrather than chasing it. Point a check at the final URL.\n\n**Every probed host must be on the probe\'s `egressAllow`.** A host that is not listed fails with\n`host not in egressAllow` \u2014 nothing leaves the function unless you named it. Add the chat host too\n(`hooks.slack.com`; `discord.com` for Discord; `chat.googleapis.com` for a Google Chat space webhook, which\ntakes `{ text }` alone \u2014 the function sends only that field to it).\n\n**What cannot be probed.** Private networks, loopback and internal hostnames are unreachable from a\nfunction by design. To watch an internal service, expose a small **public health endpoint** for it and\nguard it with the probe token:\n\n```text\nGET https://api.example.com/health x-probe-token: <probe_token>\n\u2192 200 when the service and its dependencies answer, 503 otherwise; 401 without the token\n```\n\nThe token is a per-function secret (`secret:probe_token`), so it never appears in your config or your\nrepository.\n\nThe seeded http checks land **paused** (`state: "paused"`, `enabled: false`) until their hosts are on the\nallowlist: edit the URLs, list the hosts, `vxil push`, then turn each one on with\n`{"op":"resume","check":"api-health"}` (and `website-home`) through `outage-action`. `resume` only accepts a\npaused check, sets it `up` and due now, and the next tick probes it.\n\n## Dead-man\'s-switch heartbeats\n\nSome failures make no request fail: a nightly backup that silently stopped running. A `heartbeat` check\ninverts the probe \u2014 **your job calls vxil** when it finishes, and silence is the alarm:\n\n```bash\n# the last step of the nightly backup job, wherever it runs\ncurl -s -X POST https://api.vxil.com/v1/fn/heartbeat \\\n -H "authorization: Bearer $VXIL_HEARTBEAT_KEY" -H \'content-type: application/json\' \\\n -d \'{"check":"nightly-backup"}\'\n```\n\n`VXIL_HEARTBEAT_KEY` is a **Server-class** key holding **only `functions:invoke`** (the `job-heartbeat` row in\n[the key table](#keys)), stored in that job\'s CI secret store. The\nfunction stamps `last_heartbeat_at` and makes the check due now; the probe marks the check down when the\nlast heartbeat is older than `heartbeat_grace_min` (no fetch, no egress). A check that has never beaten is\nnot judged: there is no `pending` state, so it keeps the state it has (`up` for a new check) and each probe\nrun only reschedules it, reporting it with `pending: true` in that run\'s result until the first heartbeat\narrives. The function answers `404` for an unknown check,\n`409` for a paused one or a non-heartbeat check, and `403` when it is called from a signed-in session:\nheartbeats are machine calls.\n\nA `functions:invoke` key can call any of the project\'s functions over HTTP, not only this one. Treat it\nlike any other credential, and keep `outage-action` safe on its own: it refuses a caller without a staff\nsession.\n\n## Once-per-crossing outages, paged from data\n\nThe state machine is small:\n\n- **up \u2192 down** after `failures_to_down` **consecutive** failures (2 by default). One failure is a blip.\n- **down \u2192 up** on the **first** success.\n- **degraded** when a successful answer took longer than `degraded_ms` \u2014 the component turns `degraded`\n and the chat gets one line, but no outage opens.\n\nEach tick first **claims** the check with one `If-Match` write that also moves `next_due_at` forward. A\nre-delivered or overlapping tick gets `409` there and stands down, so everything after the claim happens\nonce.\n\nOn the down-crossing the probe creates the outage under a **lock and a guard**:\n\n```json\nPOST /v1/cms/items/outages\n{ "data": { "check": "api-health", "state": "open", "severity": "major", \u2026 },\n "status": "published",\n "lock": "outage:api-health",\n "guard": { "filter": { "check": "api-health", "state": { "$in": ["open", "acked"] } }, "max": 1 } }\n```\n\nA check can therefore never have two open outages: a second create is `409 guard_failed`, which the probe\ntreats as *already open* and adopts. It then **pages the shift that covers now** \u2014 `oncall_shifts` with\n`starts_at \u2264 now < ends_at` and role `primary` \u2014 in chat once, and to each person by e-mail and in the\nin-app inbox. The page is claimed on the outage with an `if: { "paged": null }` precondition, so however\nthe tick is delivered, an outage pages at most once.\n\n**Escalation is not an engine.** In the same tick, the probe looks for outages still `open` (nobody acked)\n`ESCALATE_AFTER_MIN` (15) minutes after they opened, stamps `escalated_at` on each with a precondition, and\npages the `secondary`. Set the constant to `0` to turn it off.\n\nOn recovery the probe resolves the outage with an `if` on its state (`open` or `acked`), writes\n`duration_min`, and posts **recovered** once. If a staff member resolved it by hand while the check was\nstill failing, the probe does not reopen it until the check has recovered and failed again.\n\n## Keys\n\nMint the keys in the dashboard under **Keys \u2192 New key** (or with `vxil keys mint`, which needs a dashboard\nsession: `vxil login` first). Two of them live in the staff app, one in your jobs:\n\n| Key | Class | Scopes | Where it lives | Why |\n|---|---|---|---|---|\n| `web-auth` | Server | `auth:signin` | the staff app | signing in is how a session is obtained, so this key cannot require one |\n| `web-data` | Public / thin-client | `cms:read`, `functions:invoke` | the staff app | refused by the edge without a valid staff session |\n| `job-heartbeat` | Server | `functions:invoke` | your job\'s CI secret store, **never a browser** | heartbeats are machine calls with no session |\n\n`web-data` holds **no `cms:write`**: the staff app reads the collections directly, and every write goes\nthrough `outage-action`, which checks the person\'s permission. Because it is a thin-client key, every call\nit makes carries a signed-in session, and `heartbeat` refuses any call that carries one. **The staff app\ntherefore cannot send a heartbeat.**\n\n**Never ship a Server-class key with `functions:invoke` in a browser.** Anyone can copy a key out of a\npage. A Server-class key with no session can call `heartbeat` for any check key, as often as it likes, and\neach fake beat keeps that check `up`. A backup that stopped running would then never page anyone, which\ndefeats the heartbeat check. The heartbeat\'s only authentication is the key: the function accepts any caller\nholding `functions:invoke` *without* a session. Keep `job-heartbeat` in the job\'s secret store, give it no\nother scope, and rotate it like any other credential.\n\n`cms:read` on `web-data` lets **anyone who can sign in** read the staff collections, including check URLs,\noutages and the on-call rota. Signing in is fenced only by `allowedDomains`, so that means anyone at your\ncompany domain, not just members of the team org. Org roles gate the actions, not the reads. If the rota or\nthe internal URLs should be narrower than that, assign the identity-provider app to the on-call group only.\n\n## Staff actions\n\nYour staff app calls `outage-action` with the `web-data` key plus the signed-in staff member\'s session:\n\n```json\nPOST /v1/fn/outage-action\n{ "op": "ack", "outage_id": "\u2026" }\n{ "op": "resolve", "outage_id": "\u2026" }\n{ "op": "post_update", "outage_id": "\u2026", "kind": "identified", "body": "\u2026", "publish": true,\n "component_status": "partial_outage" }\n{ "op": "pause", "check": "api-health" }\n{ "op": "resume", "check": "api-health" }\n{ "op": "publish_component", "key": "api" }\n```\n\n`ack`, `resolve` and `post_update` need **`outages.manage`**; `pause`, `resume` and `publish_component` need\n**`checks.manage`**. The function re-checks the permission **live** on every call\n(`GET /v1/orgs/session-claims`), not from the role in the session, so a member you remove loses the buttons\nat once. Every state change is a conditional write and the `outage_stage` hook is the authority, so two\npeople pressing *Ack* at once produce one ack and one `409`.\n\n## The public status page\n\n`status_components` and `outage_updates` are `public: true`. Their **published** rows are readable with **no\nAPI key**, edge-cached, from any front end:\n\n```bash\ncurl "https://api.vxil.com/v1/cms/public/<TENANT_ID>/status_components?sort=sort"\ncurl "https://api.vxil.com/v1/cms/public/<TENANT_ID>/outage_updates?sort=-at&limit=20"\n```\n\n- **Only published rows are served.** A `post_update` with `publish: false` \u2014 and every `kind: "note"` \u2014\n is written as a **draft**: staff see it, the public never does. Notes are where internal detail belongs.\n- **Never put an internal URL, hostname or customer name in a published update.** The body is public the\n moment it is published.\n- The `author` of an update is role-gated (`readRoles`), so the public lane never serves your staff\'s ids.\n- Seeds land as drafts. Publish the two seeded components once:\n `{"op":"publish_component","key":"api"}` and `{"op":"publish_component","key":"website"}`.\n- The checks, their URLs, the outages and the shifts are **not** public. A component\'s status moves when\n the probe sees a crossing, or when staff publish an update with `component_status`.\n\n## Uptime counters\n\nOne `uptime_hourly` row per check per hour. The first sample of an hour creates the row (`point_key` is\nunique, so a racing create is a `409`, which turns into an increment on the existing row); every later\nsample is one atomic write:\n\n```json\nPATCH /v1/cms/items/uptime_hourly/<id>\n{ "$inc": { "ok": 1, "total": 1, "latency_sum": 182.4 } }\n```\n\n`$inc` is a single conditional statement \u2014 no read first, no lost update. `latency_max` is written only\nwhen a sample beats the hour\'s maximum.\n\n```\nuptime % = 100 \xD7 \u03A3 ok / \u03A3 total over the rows in the window\navg latency = \u03A3 latency_sum / \u03A3 ok (successful samples only)\n```\n\n`daily-report` (08:00 UTC) computes the first one for every check in **one server-side aggregate** \u2014\ngroup by `check`, sum `ok` and `total`, a 1-day window on `ts` \u2014 and e-mails it, worst first, with the\noutages opened that day, to whoever is on shift. Function crons run in UTC; for local time, enqueue the\nreport from a jobs schedule that carries a timezone.\n\n## Compose it\n\n- **alerts-to-slack** watches the *platform\'s own* failure events \u2014 this blueprint\'s functions failing,\n dead-lettered notifications, a quarantined function. Run both: one watches your services, the other\n watches the monitor.\n- **payments-heartbeat** is the same once-per-crossing pattern applied to silence from a payments\n integration\'s provider.\n- An **ops dashboard** in the same project can read `checks`, `outages` and `uptime_hourly` for its tiles \u2014\n the collection names here are specific to monitoring, so they do not clash.\n\n`vxil push` writes each feature\'s config as a whole, so to combine blueprints in one project, copy the\ncollections, hooks and functions of one into the other\'s `vxil.config.ts` and push that file.\n\n## Cadence and counts\n\n| | Free | Developer and above |\n|---|---|---|\n| probe cron | `*/15 * * * *` | `* * * * *` |\n| functions / crons used | 4 of 5 / 2 of 3 | 4 / 2 |\n| checks per tick | 40 (oldest due first) | 40 \u2014 raise `MAX_PER_TICK` with care: each check costs a probe and 2\u20133 writes |\n| finest interval | 15 minutes | 1 minute |\n\nA tick probes up to 40 checks; the rest wait one tick. With the default one-minute interval that is about\n40 checks per project; at `interval_min: 5` it is about 200.\n\n## What to learn from this\n\n- **Deny-by-default egress is a feature of a monitor.** The probe can reach exactly the hosts you listed \u2014\n and the public health endpoint plus a token is how an internal service joins the list safely.\n- **A crossing is a write that can only succeed once.** A lock + guard, an `If-Match` claim, and an `if`\n precondition turn an at-least-once cron into exactly one outage, one page and one "recovered".\n- **On-call is data.** Who to page is a filtered read of `oncall_shifts`, and escalation is the same cron\n asking one more question.\n- **Public by publishing.** A keyless status page needs no server: two public collections, and a draft\n that stays internal until someone chooses to publish it.\n',
|
|
18548
|
+
"readme": '# Service monitor (ops)\n\nUptime and health checks for your team\'s **public endpoints**, **dead-man\'s-switch heartbeats** from your\nown scheduled jobs, outages that open and close **once per state crossing**, the on-call shift **paged from\ndata** in chat, e-mail and the in-app inbox, staff who acknowledge, resolve and post updates, and a\n**keyless public status page** that shows component status and only the updates you chose to publish.\n\nIt is four functions and six collections. There is no monitoring engine, no escalation engine and no\nconnector in it: a probe is a cron function behind a deny-by-default egress allowlist, an outage is a row\nwhose creation is guarded, a page is a read of who is on shift right now, and escalation is one more\nfilter in the same cron.\n\n```bash\nmkdir service-monitor && cd service-monitor && vxil init --template service-monitor # init scaffolds into the current directory\nvxil quickstart --env staging --no-push --invite <code> # or `vxil link <slug> --env staging`\nvxil projects workload <slug> staging # Free plan: functions deploy on staging/development\nprintf \'%s\' "$SLACK_URL" | vxil secrets set functions/slack_webhook_url\nprintf \'%s\' "$PROBE_TOKEN" | vxil secrets set functions/probe_token\nprintf \'%s\' "$OIDC_SECRET" | vxil secrets set auth/oidc_client_secret\nvxil push # collections, hooks, auth, orgs, the four functions\nvxil seed # 2 components, 2 http checks (paused), 1 heartbeat\n```\n\nThen, once:\n\n```bash\n# the two custom roles (rows, not config \u2014 `viewer` and `owner` are built in)\nvxil api POST /v1/orgs/roles --data \'{"role_key":"oncall","name":"On call","permissions":["outages.manage"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"monitor-admin","name":"Monitor admin","permissions":["outages.manage","checks.manage"]}\'\n# the team org (its owner holds every permission), then each staff member with a role\n# \u2014 a member\'s user id exists after their first sign-in\nvxil api POST /v1/orgs --data \'{"slug":"platform","name":"Platform team","owner_user_id":"<your user id>"}\'\nvxil api POST /v1/orgs/<org_id>/members --data \'{"user_id":"<user id>","role":"oncall"}\'\n# who is on call, as data\nvxil api POST /v1/cms/items/oncall_shifts --data \'{"status":"published","data":{"user":"<user id>","role":"primary","starts_at":"2026-10-06T08:00:00Z","ends_at":"2026-10-13T08:00:00Z"}}\'\n```\n\nEdit `issuer`, `clientId` and `allowedDomains` in `vxil.config.ts` to your identity provider before you push.\n\n**Both probe secrets must be set before the probe can run**: a function whose declared secret has no value\nis refused at invoke (`409 secret_missing`), so the cron would fail every tick. No chat? Delete\n`\'secret:slack_webhook_url\'` from the probe\'s `secrets` (the probe then pages by e-mail and inbox only). No\ntoken-guarded endpoint yet? Set `probe_token` to any random value, or delete it the same way.\n\n> **Plan note.** Free deploys functions to `staging` and `development` projects, caps a project at 5\n> functions and 3 crons, and runs a cron every 15 minutes at the fastest. This blueprint uses 4 functions\n> and 2 crons, so it fits \u2014 change the probe\'s `schedule` to `*/15 * * * *` on Free (the config comment\n> says so), or the deploy answers `402 plan_limit`. Developer and above run it every minute.\n\n## The collections\n\n| Collection | What it holds | Who reads it |\n|---|---|---|\n| `checks` | what to watch: `key`, `kind` (`http` \xB7 `keyword` \xB7 `heartbeat`), `url`, expected status/text, timeout, `degraded_ms`, `interval_min`, `failures_to_down`, and the live state (`up` \xB7 `degraded` \xB7 `down` \xB7 `paused`) | staff |\n| `uptime_hourly` | one row per check per hour: `ok`, `total`, `latency_sum`, `latency_max` | staff, the daily report |\n| `outages` | one row per down-crossing: `open` \u2192 `acked` \u2192 `resolved`, who acked, duration, who was paged | staff |\n| `outage_updates` \u2B22 | the updates staff write; **only published ones are public** | everyone (published only) |\n| `status_components` \u2B22 | the components on your status page and their status | everyone (published only) |\n| `oncall_shifts` | who is on call, `primary` or `secondary`, from when to when | staff, the probe |\n\nThree Lane-A hooks guard the data in the same write: `outage_stage` (open \u2192 acked \u2192 resolved, resolved is\nfinal), `check_kind` and `check_url` (an http or keyword check needs an `https://` url), and `uptime_key`,\nwhich derives `point_key = check|bucket` server-side on a field declared `unique` \u2014 so one hour can never\nhave two rows for the same check.\n\n## Probes: a cron function with deny-by-default egress\n\n`probe` runs every minute. It reads the due checks \u2014 `enabled`, `next_due_at \u2264 now`, oldest first, at most\n`MAX_PER_TICK` (40) \u2014 and probes them in parallel:\n\n- **http** \u2014 `GET url`, with the check\'s `timeout_ms` (\u2264 15 s), `redirect: \'manual\'`, and the\n `x-probe-token` header when the check sets `send_probe_token`. Up when the status equals `expect_status`\n (200 by default).\n- **keyword** \u2014 the same, and the body must contain `expect_contains`.\n- **heartbeat** \u2014 no fetch at all (below).\n\nA **redirect is a failure** with reason `redirect`. The probe never follows one: a public host can redirect\nto a private one, so the platform\'s egress guard refuses to follow redirects, and the probe reports it\nrather than chasing it. Point a check at the final URL.\n\n**Every probed host must be on the probe\'s `egressAllow`.** A host that is not listed fails with\n`host not in egressAllow` \u2014 nothing leaves the function unless you named it. Add the chat host too\n(`hooks.slack.com`; `discord.com` for Discord; `chat.googleapis.com` for a Google Chat space webhook, which\ntakes `{ text }` alone \u2014 the function sends only that field to it).\n\n**What cannot be probed.** Private networks, loopback and internal hostnames are unreachable from a\nfunction by design. To watch an internal service, expose a small **public health endpoint** for it and\nguard it with the probe token:\n\n```text\nGET https://api.example.com/health x-probe-token: <probe_token>\n\u2192 200 when the service and its dependencies answer, 503 otherwise; 401 without the token\n```\n\nThe token is a per-function secret (`secret:probe_token`), so it never appears in your config or your\nrepository.\n\nThe seeded http checks land **paused** (`state: "paused"`, `enabled: false`) until their hosts are on the\nallowlist: edit the URLs, list the hosts, `vxil push`, then turn each one on with\n`{"op":"resume","check":"api-health"}` (and `website-home`) through `outage-action`. `resume` only accepts a\npaused check, sets it `up` and due now, and the next tick probes it.\n\n## Dead-man\'s-switch heartbeats\n\nSome failures make no request fail: a nightly backup that silently stopped running. A `heartbeat` check\ninverts the probe \u2014 **your job calls vxil** when it finishes, and silence is the alarm:\n\n```bash\n# the last step of the nightly backup job, wherever it runs\ncurl -s -X POST https://api.vxil.com/v1/fn/heartbeat \\\n -H "authorization: Bearer $VXIL_HEARTBEAT_KEY" -H \'content-type: application/json\' \\\n -d \'{"check":"nightly-backup"}\'\n```\n\n`VXIL_HEARTBEAT_KEY` is a **Server-class** key holding **only `functions:invoke`** (the `job-heartbeat` row in\n[the key table](#keys)), stored in that job\'s CI secret store. The\nfunction stamps `last_heartbeat_at` and makes the check due now; the probe marks the check down when the\nlast heartbeat is older than `heartbeat_grace_min` (no fetch, no egress). A check that has never beaten is\nnot judged: there is no `pending` state, so it keeps the state it has (`up` for a new check) and each probe\nrun only reschedules it, reporting it with `pending: true` in that run\'s result until the first heartbeat\narrives. The function answers `404` for an unknown check,\n`409` for a paused one or a non-heartbeat check, and `403` when it is called from a signed-in session:\nheartbeats are machine calls.\n\nA `functions:invoke` key can call any of the project\'s functions over HTTP, not only this one. Treat it\nlike any other credential, and keep `outage-action` safe on its own: it refuses a caller without a staff\nsession.\n\n## Once-per-crossing outages, paged from data\n\nThe state machine is small:\n\n- **up \u2192 down** after `failures_to_down` **consecutive** failures (2 by default). One failure is a blip.\n- **down \u2192 up** on the **first** success.\n- **degraded** when a successful answer took longer than `degraded_ms` \u2014 the component turns `degraded`\n and the chat gets one line, but no outage opens.\n\nEach tick first **claims** the check with one `If-Match` write that also moves `next_due_at` forward. A\nre-delivered or overlapping tick gets `409` there and stands down, so everything after the claim happens\nonce.\n\nOn the down-crossing the probe creates the outage under a **lock and a guard**:\n\n```json\nPOST /v1/cms/items/outages\n{ "data": { "check": "api-health", "state": "open", "severity": "major", \u2026 },\n "status": "published",\n "lock": "outage:api-health",\n "guard": { "filter": { "check": "api-health", "state": { "$in": ["open", "acked"] } }, "max": 1 } }\n```\n\nA check can therefore never have two open outages: a second create is `409 guard_failed`, which the probe\ntreats as *already open* and adopts. It then **pages the shift that covers now** \u2014 `oncall_shifts` with\n`starts_at \u2264 now < ends_at` and role `primary` \u2014 in chat once, and to each person by e-mail and in the\nin-app inbox. The page is claimed on the outage with an `if: { "paged": null }` precondition, so however\nthe tick is delivered, an outage pages at most once.\n\n**Escalation is not an engine.** In the same tick, the probe looks for outages still `open` (nobody acked)\n`ESCALATE_AFTER_MIN` (15) minutes after they opened, stamps `escalated_at` on each with a precondition, and\npages the `secondary`. Set the constant to `0` to turn it off.\n\nOn recovery the probe resolves the outage with an `if` on its state (`open` or `acked`), writes\n`duration_min`, and posts **recovered** once. If a staff member resolved it by hand while the check was\nstill failing, the probe does not reopen it until the check has recovered and failed again.\n\n## Keys\n\nMint the keys in the dashboard under **Keys \u2192 New key** (or with `vxil keys mint`, which needs a dashboard\nsession: `vxil login` first). Two of them live in the staff app, one in your jobs:\n\n| Key | Class | Scopes | Where it lives | Why |\n|---|---|---|---|---|\n| `web-auth` | Server | `auth:signin` | the staff app | signing in is how a session is obtained, so this key cannot require one |\n| `web-data` | Public / thin-client | `cms:read`, `functions:invoke` | the staff app | refused by the edge without a valid staff session |\n| `job-heartbeat` | Server | `functions:invoke` | your job\'s CI secret store, **never a browser** | heartbeats are machine calls with no session |\n\n`web-data` holds **no `cms:write`**: the staff app reads the collections directly, and every write goes\nthrough `outage-action`, which checks the person\'s permission. Because it is a thin-client key, every call\nit makes carries a signed-in session, and `heartbeat` refuses any call that carries one. **The staff app\ntherefore cannot send a heartbeat.**\n\n**Never ship a Server-class key with `functions:invoke` in a browser.** Anyone can copy a key out of a\npage. A Server-class key with no session can call `heartbeat` for any check key, as often as it likes, and\neach fake beat keeps that check `up`. A backup that stopped running would then never page anyone, which\ndefeats the heartbeat check. The heartbeat\'s only authentication is the key: the function accepts any caller\nholding `functions:invoke` *without* a session. Keep `job-heartbeat` in the job\'s secret store, give it no\nother scope, and rotate it like any other credential.\n\n`cms:read` on `web-data` lets **anyone who can sign in** read the staff collections, including check URLs,\noutages and the on-call rota. Signing in is fenced only by `allowedDomains`, so that means anyone at your\ncompany domain, not just members of the team org. Org roles gate the actions, not the reads. If the rota or\nthe internal URLs should be narrower than that, assign the identity-provider app to the on-call group only.\n\n## Staff actions\n\nYour staff app calls `outage-action` with the `web-data` key plus the signed-in staff member\'s session:\n\n```json\nPOST /v1/fn/outage-action\n{ "op": "ack", "outage_id": "\u2026" }\n{ "op": "resolve", "outage_id": "\u2026" }\n{ "op": "post_update", "outage_id": "\u2026", "kind": "identified", "body": "\u2026", "publish": true,\n "component_status": "partial_outage" }\n{ "op": "pause", "check": "api-health" }\n{ "op": "resume", "check": "api-health" }\n{ "op": "publish_component", "key": "api" }\n```\n\n`ack`, `resolve` and `post_update` need **`outages.manage`**; `pause`, `resume` and `publish_component` need\n**`checks.manage`**. The function re-checks the permission **live** on every call\n(`GET /v1/orgs/session-claims`), not from the role in the session, so a member you remove loses the buttons\nat once. Every state change is a conditional write and the `outage_stage` hook is the authority, so two\npeople pressing *Ack* at once produce one ack and one `409`.\n\n## The public status page\n\n`status_components` and `outage_updates` are `public: true`. Their **published** rows are readable with **no\nAPI key**, edge-cached, from any front end:\n\n```bash\ncurl "https://api.vxil.com/v1/cms/public/<TENANT_ID>/status_components?sort=sort"\ncurl "https://api.vxil.com/v1/cms/public/<TENANT_ID>/outage_updates?sort=-at&limit=20"\n```\n\n- **Only published rows are served.** A `post_update` with `publish: false` \u2014 and every `kind: "note"` \u2014\n is written as a **draft**: staff see it, the public never does. Notes are where internal detail belongs.\n- **Never put an internal URL, hostname or customer name in a published update.** The body is public the\n moment it is published.\n- The `author` of an update is role-gated (`readRoles`), so the public lane never serves your staff\'s ids.\n- Seeds land as drafts. Publish the two seeded components once:\n `{"op":"publish_component","key":"api"}` and `{"op":"publish_component","key":"website"}`.\n- The checks, their URLs, the outages and the shifts are **not** public. A component\'s status moves when\n the probe sees a crossing, or when staff publish an update with `component_status`.\n\n## Uptime counters\n\nOne `uptime_hourly` row per check per hour. The first sample of an hour creates the row (`point_key` is\nunique, so a racing create is a `409`, which turns into an increment on the existing row); every later\nsample is one atomic write:\n\n```json\nPATCH /v1/cms/items/uptime_hourly/<id>\n{ "$inc": { "ok": 1, "total": 1, "latency_sum": 182.4 } }\n```\n\n`$inc` is a single conditional statement \u2014 no read first, no lost update. `latency_max` is written only\nwhen a sample beats the hour\'s maximum.\n\n```\nuptime % = 100 \xD7 \u03A3 ok / \u03A3 total over the rows in the window\navg latency = \u03A3 latency_sum / \u03A3 ok (successful samples only)\n```\n\n`daily-report` (08:00 UTC) computes the first one for every check in **one server-side aggregate** \u2014\ngroup by `check`, sum `ok` and `total`, a 1-day window on `ts` \u2014 and e-mails it, worst first, with the\noutages opened that day, to whoever is on shift. Function crons run in UTC; for local time, enqueue the\nreport from a jobs schedule that carries a timezone.\n\n## Compose it\n\n- **alerts-to-slack** watches the *platform\'s own* failure events \u2014 this blueprint\'s functions failing,\n dead-lettered notifications, a quarantined function. Run both: one watches your services, the other\n watches the monitor.\n- **payments-heartbeat** is the same once-per-crossing pattern applied to silence from a payments\n integration\'s provider.\n- An **ops dashboard** in the same project can read `checks`, `outages` and `uptime_hourly` for its tiles \u2014\n the collection names here are specific to monitoring, so they do not clash.\n\n`vxil push` writes each feature\'s config as a whole, so to combine blueprints in one project, copy the\ncollections, hooks and functions of one into the other\'s `vxil.config.ts` and push that file.\n\n## Cadence and counts\n\n| | Free | Developer and above |\n|---|---|---|\n| probe cron | `*/15 * * * *` | `* * * * *` |\n| functions / crons used | 4 of 5 / 2 of 3 | 4 / 2 |\n| checks per tick | 40 (oldest due first) | 40 \u2014 raise `MAX_PER_TICK` with care: each check costs a probe and 2\u20133 writes |\n| finest interval | 15 minutes | 1 minute |\n\nA tick probes up to 40 checks; the rest wait one tick. With the default one-minute interval that is about\n40 checks per project; at `interval_min: 5` it is about 200.\n\n## What to learn from this\n\n- **Deny-by-default egress is a feature of a monitor.** The probe can reach exactly the hosts you listed \u2014\n and the public health endpoint plus a token is how an internal service joins the list safely.\n- **A crossing is a write that can only succeed once.** A lock + guard, an `If-Match` claim, and an `if`\n precondition turn an at-least-once cron into exactly one outage, one page and one "recovered".\n- **On-call is data.** Who to page is a filtered read of `oncall_shifts`, and escalation is the same cron\n asking one more question.\n- **Public by publishing.** A keyless status page needs no server: two public collections, and a draft\n that stays internal until someone chooses to publish it.\n',
|
|
18507
18549
|
"functions": {
|
|
18508
18550
|
"daily-report.ts": "// daily-report.ts \u2014 THE MORNING E-MAIL (a vxil function, cron trigger).\n//\n// Trigger: cron `0 8 * * *`. Function crons run in UTC; if your team wants it\n// at 08:00 local time, enqueue it from a jobs schedule that carries a timezone.\n//\n// ONE server-side aggregate over uptime_hourly \u2014 group by check, sum ok, sum\n// total, over the last day (the `ts` window) \u2014 gives every check's uptime:\n//\n// uptime % = 100 \xD7 \u03A3 ok / \u03A3 total (per check, last 24 h)\n//\n// plus the outages opened in the same window, e-mailed to whoever is on shift\n// now. A check with no samples in the window is simply absent from the table.\n\n// cron-walk: single-read \u2014 one bounded server-side aggregate (\u2264500 groups) and one bounded outage list are the whole job; nothing to page.\n\nimport type { CronFunctionEnvelope } from '@vxil/sdk';\n\nconst MAX_OUTAGES_LISTED = 50;\nconst MAX_LINES = 60; // checks listed in the e-mail body\n\ninterface Group { key: { check?: string }; ok?: number; total?: number }\ninterface Row<T> { item_id: string; data: T }\ninterface OutageData { check: string; state?: string; title?: string; opened_at?: string; duration_min?: number }\ninterface ShiftData { user: string; email?: string }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as CronFunctionEnvelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n const notify = env.scoped_jwts?.notifications;\n if (!cms || !notify) return Response.json({ error: 'missing cms or notifications scope' }, { status: 403 });\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n const now = new Date();\n const nowIso = now.toISOString();\n const since = new Date(now.getTime() - 86_400_000).toISOString();\n\n // 1. the aggregate: sum needs ok/total on n-slots (n1, n2); the 1-day window rides ts (t1)\n const agg = await fetch(`${base}/v1/cms/items/uptime_hourly/aggregate`, {\n method: 'POST', headers: H,\n body: JSON.stringify({\n aggregates: [\n { fn: 'sum', field: 'ok', as: 'ok' },\n { fn: 'sum', field: 'total', as: 'total' },\n ],\n groupBy: ['check'],\n window: { field: 'ts', sinceDays: 1 },\n limit: 500,\n }),\n });\n if (!agg.ok) return Response.json({ error: 'aggregate_failed', status: agg.status }, { status: 502 });\n const groups = ((await agg.json()) as { data?: { groups?: Group[] } }).data?.groups ?? [];\n\n // 2. the outages opened in the same window\n const of = encodeURIComponent(JSON.stringify({ opened_at: { $gte: since } }));\n const ol = await fetch(`${base}/v1/cms/items/outages?filter=${of}&sort=-opened_at&limit=${MAX_OUTAGES_LISTED}`, { headers: H });\n const outages = ol.ok ? ((await ol.json()) as { data?: { items?: Row<OutageData>[] } }).data?.items ?? [] : [];\n\n // 3. who gets it: everyone on shift right now\n const sf = encodeURIComponent(JSON.stringify({ starts_at: { $lte: nowIso }, ends_at: { $gt: nowIso } }));\n const sl = await fetch(`${base}/v1/cms/items/oncall_shifts?filter=${sf}&limit=20`, { headers: H });\n const shifts = sl.ok ? ((await sl.json()) as { data?: { items?: Row<ShiftData>[] } }).data?.items ?? [] : [];\n if (shifts.length === 0) return Response.json({ checks: groups.length, outages: outages.length, sent: 0, reason: 'nobody on shift' });\n\n const rows = groups\n .filter((g) => g.key.check && (g.total ?? 0) > 0)\n .map((g) => ({ check: g.key.check!, pct: (100 * (g.ok ?? 0)) / (g.total ?? 1), total: g.total ?? 0 }))\n .sort((a, b) => a.pct - b.pct); // worst first\n const lines = rows.slice(0, MAX_LINES).map((r) => `${r.check}: ${r.pct.toFixed(2)} % of ${r.total} checks`);\n if (rows.length > MAX_LINES) lines.push(`\u2026 and ${rows.length - MAX_LINES} more`);\n const outageLines = outages.map((o) => `${o.data.title ?? o.data.check} \u2014 ${o.data.state ?? 'open'}${o.data.duration_min !== undefined ? ` (${o.data.duration_min} min)` : ''}`);\n const paragraph = [\n `Uptime, last 24 h (worst first): ${lines.length ? lines.join(' \xB7 ') : 'no samples'}.`,\n `Outages opened: ${outageLines.length ? outageLines.join(' \xB7 ') : 'none'}.`,\n ].join(' ');\n\n // one report per person per day, however often this tick is delivered\n const day = nowIso.slice(0, 10);\n let sent = 0;\n for (const s of dedupe(shifts.map((r) => r.data))) {\n const res = await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: { authorization: `Bearer ${notify}`, 'content-type': 'application/json', 'idempotency-key': `daily-report:${day}:${s.user}` },\n body: JSON.stringify({\n user_id: s.user,\n ...(s.email ? { to_email: s.email } : {}),\n template: 'transactional',\n channel: 'email',\n data: { subject: `Service monitor \u2014 daily report ${day}`, paragraph },\n }),\n }).catch(() => null);\n if (res?.ok) sent++;\n }\n return Response.json({ checks: rows.length, outages: outages.length, sent });\n },\n};\n\nfunction dedupe(shifts: ShiftData[]): ShiftData[] {\n const seen = new Set<string>();\n return shifts.filter((s) => (seen.has(s.user) ? false : (seen.add(s.user), true)));\n}\n",
|
|
18509
18551
|
"heartbeat.ts": "// heartbeat.ts \u2014 THE DEAD-MAN'S SWITCH (a vxil function, http trigger).\n//\n// Your own scheduled jobs (a backup, an export, a nightly sync \u2014 wherever they\n// run) call this when they FINISH:\n//\n// curl -s -X POST https://api.vxil.com/v1/fn/heartbeat \\\n// -H \"authorization: Bearer $VXIL_HEARTBEAT_KEY\" -H 'content-type: application/json' \\\n// -d '{\"check\":\"nightly-backup\"}'\n//\n// with a SERVER-class key holding ONLY `functions:invoke`, kept in that job's CI\n// secret store and never shipped to a browser: the key is this endpoint's only\n// authentication, so whoever holds it can keep a dead job's check `up`. The\n// staff app's thin-client key always carries a session, which is refused here. It stamps `last_heartbeat_at` on the check and makes the check due now,\n// so the next probe tick (\u2264 1 minute) judges it \u2014 a check that was down\n// recovers on its first beat. The PROBE is what notices silence: a check whose\n// last heartbeat is older than its grace goes down there, with no fetch.\n//\n// 404 \u2014 no such check \xB7 409 \u2014 not a heartbeat check, or paused/disabled\n// 403 \u2014 called from a signed-in session: heartbeats are machine calls\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\ntype Env = HttpFunctionEnvelope<{ check?: string }>;\ninterface CheckData { key: string; kind?: string; state?: string; enabled?: boolean }\ninterface Row { item_id: string; data: CheckData }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n if (!cms) return fail(403, 'missing_scope', 'the function has no cms token');\n if (env.end_user) return fail(403, 'machine_only', 'a heartbeat comes from a job with a functions:invoke key, not from a signed-in session');\n const key = env.payload?.check;\n if (typeof key !== 'string' || !/^[A-Za-z0-9._:-]{1,120}$/.test(key)) {\n return fail(422, 'bad_check', 'send {\"check\":\"<check key>\"}');\n }\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n\n const filter = encodeURIComponent(JSON.stringify({ key }));\n const found = await fetch(`${base}/v1/cms/items/checks?filter=${filter}&limit=1`, { headers: H });\n if (!found.ok) return fail(502, 'lookup_failed', `checks read ${found.status}`);\n const row = ((await found.json()) as { data?: { items?: Row[] } }).data?.items?.[0];\n if (!row) return fail(404, 'unknown_check', `no check '${key}'`);\n if (row.data.kind !== 'heartbeat') return fail(409, 'not_a_heartbeat', `'${key}' is a ${row.data.kind} check`);\n if (row.data.state === 'paused' || row.data.enabled === false) return fail(409, 'paused', `'${key}' is paused`);\n\n const now = new Date().toISOString();\n const res = await fetch(`${base}/v1/cms/items/checks/${row.item_id}`, {\n method: 'PATCH',\n headers: H,\n // `if`: a check paused between the read and this write stays paused\n body: JSON.stringify({ data: { last_heartbeat_at: now, next_due_at: now }, if: { state: { $ne: 'paused' } } }),\n });\n if (res.status === 409) return fail(409, 'paused', `'${key}' is paused`);\n if (!res.ok) return fail(502, 'write_failed', `checks write ${res.status}`);\n return Response.json({ check: key, beat_at: now });\n },\n};\n\nconst fail = (status: number, code: string, message: string) => Response.json({ error: { code, message } }, { status });\n",
|
|
@@ -18560,7 +18602,7 @@ export default defineConfig({
|
|
|
18560
18602
|
"vxil_jobs_key"
|
|
18561
18603
|
],
|
|
18562
18604
|
"configSrc": "import { defineConfig } from '@vxil/config';\n\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n// \"Render Farm\" \u2014 long renders and transcodes (minutes, ffmpeg, a headless\n// browser) run on a runtime YOU rent: Trigger.dev, Modal, a container on your\n// own cloud account. vxil keeps the four things that must not be lost while\n// that runtime works: the credits held for the render, the deadline, the\n// signed completion callback, and the status row the app watches.\n//\n// request-render (your function, end-user mode)\n// \u2192 creates-or-finds the user's `renders` row (unique render_key)\n// \u2192 POST /v1/jobs/generation in WEBHOOK mode: your render endpoint,\n// credits HELD, a deadline, the status mirrored onto the row\n// vxil's generation lane\n// \u2192 POSTs your render endpoint { render_id, user_id, composition, props,\n// payload, callback_url } with your bearer token\n// your runtime (README: Trigger.dev, or your own containers)\n// \u2192 acks within 20 s, renders, POSTs { status: 'processing', \u2026 } and then\n// { status: 'completed', output_url, duration_s, \u2026 } to callback_url\n// vxil\n// \u2192 completed: credits COMMIT, every callback key lands on the row\n// \u2192 failed / no answer by the deadline: credits REFUNDED, row says failed\n// \u2192 job.generation.completed | failed \u2192 notify-ready tells the owner\n// redrive-pending (cron, every minute)\n// \u2192 starts renders that waited at the concurrency cap (429 \u2014 the app was\n// told `queued: true`), same key; tells the owner if one never starts\n//\n// No container tier and no workflow engine inside vxil: the runtime is yours,\n// the bookkeeping is vxil's. \"Credits\" are usage units on the deterministic\n// `mock` payments integration \u2014 not money; vxil is never in the flow of funds.\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\nexport default defineConfig({\n env: 'staging',\n\n features: {\n jobs: {\n enabled: true,\n generation: {\n // how many renders may be in flight at once for this project\n maxConcurrent: 20,\n // a render that never calls back fails (and refunds) after 30 minutes\u2026\n defaultTimeoutMs: 1_800_000,\n // \u2026and no render may ask for more than the platform ceiling, one hour\n maxTimeoutMs: 3_600_000,\n // one render never holds more than 50 credits\n maxReserveCredits: 50,\n // all of this project's in-flight renders together hold at most 5,000\n maxOutstandingReserveCredits: 5_000,\n },\n },\n\n payments: {\n enabled: true,\n provider: 'mock',\n defaults: { currency: 'usd' },\n ledger: {\n productMap: {\n render_pack_100: { creditType: 'render_credits', amount: 100, period: 'once' },\n },\n // no subscription tiers in this blueprint \u2014 credits come from packs\n tierMap: {},\n // a render that fails, times out or is cancelled gives its credits back\n autoRefundOnJobFailure: true,\n },\n },\n\n cms: {\n // a render row is live the moment it is written\n draftPublish: false,\n // in end-user mode a signed-in user sees only the renders they own\n strictEndUserScope: true,\n // Lane-A hook (guide ch. 7): render_key IS owner + ':' + request_key,\n // server-enforced, so one user's request_key can never collide with \u2014\n // or block \u2014 another user's.\n hooks: {\n render_key_shape: {\n collection: 'renders',\n event: 'beforeWrite',\n kind: 'validate',\n expr: \"item.render_key == concat(item.owner, ':', item.request_key)\",\n message: \"render_key must be owner + ':' + request_key\",\n },\n },\n },\n\n notifications: { provider: 'mock', fromEmail: 'renders@render-farm.example' },\n functions: { enabled: true },\n\n // the README's keyless-container coordinator uploads each finished output\n // into this project's files (`output_file` below). The per-object ceiling\n // the upload-url call pre-checks is `maxObjectBytes` \u2014 100 MB by default,\n // which fits the coordinator's buffered upload (\"tens of MB\"); raise it\n // (up to 5 GB) for long or high-bitrate renders. Not using the coordinator?\n // Remove this and the `output_file` field.\n files: { enabled: true },\n },\n\n cms: {\n collections: {\n renders: {\n singular: 'render',\n ownerField: 'owner',\n fields: {\n // THE DEDUPE ANCHOR: a double tap or a retried request is a 409 that\n // request-render reads back \u2014 and the same key is the generation\n // run's idempotency_key, so the render endpoint is asked once.\n render_key: { type: 'string', required: true, unique: true, indexSlot: 's1' },\n request_key: { type: 'string', required: true },\n owner: { type: 'string', indexSlot: 's2' },\n // which composition / preset your runtime renders (your vocabulary)\n composition: { type: 'string', required: true, indexSlot: 's3' },\n // written by vxil's status mirror: pending \u2192 processing \u2192 completed | failed\n status: { type: 'string', indexSlot: 's4' },\n credits: { type: 'int', indexSlot: 'n1' },\n created_at: { type: 'datetime', indexSlot: 't1' },\n props: { type: 'json' },\n run_id: { type: 'text' },\n // \u2500\u2500 keys your runtime sends back. EVERY key of the completion body is\n // written onto this row, so each one must be a declared field (a\n // write naming an unknown field is refused: vxil then writes the\n // status word alone and the run reports `mirror_error` with the\n // keys it dropped \u2014 declare the field so its value lands too).\n output_url: { type: 'text' },\n // the files-feature object id, when a coordinator uploads the output\n // into this project's files (README \"Keys stay out of the container\")\n output_file: { type: 'file' },\n duration_s: { type: 'float' },\n // progress keys: a `processing` ping carries them onto the row\n // (request-render's status_mirror.progress_fields \u2014 progress 0..100,\n // a float: the run keeps 2 decimals,\n // stage \u2264 64 chars, message \u2264 200), and the completion body writes\n // them too (progress: 100, stage: 'done').\n progress: { type: 'float', validation: { min: 0, max: 100 } },\n stage: { type: 'string' },\n message: { type: 'text' },\n // written by notify-ready from job.generation.failed (and by\n // redrive-pending when a render never got a run)\n error: { type: 'text' },\n // written by redrive-pending: how often it tried to start this render,\n // and when it last did. The re-driver's read needs no new index \u2014\n // `status` (s4) and `created_at` (t1) are slots; `run_id: null` is a\n // residual test over the rows they pick.\n redrive_attempts: { type: 'int' },\n redriven_at: { type: 'datetime' },\n },\n },\n },\n },\n\n functions: {\n // Starts ONE render for the signed-in user. Invoke it in END-USER mode\n // (with the user's session): the held credits are forced onto that user,\n // and the row is theirs.\n 'request-render': {\n entry: './functions/request-render.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write', 'jobs:write'],\n // render_url: your render endpoint (https). render_token: the bearer\n // token that endpoint checks. Both ride the generation run; vxil's\n // generation lane \u2014 not this function \u2014 calls the endpoint.\n secrets: ['secret:render_url', 'secret:render_token'],\n egressAllow: [],\n signature: {\n input: { composition: 'string', request_key: 'string', props: 'json?' },\n output:\n '{ item_id: string; run_id: string; credits: number }'\n + ' | { duplicate: true; request_key: string; item_id: string; run_id: string | null; status: string }',\n },\n },\n\n // THE BACKLOG RE-DRIVER. At the generation cap request-render answers 429\n // and the row waits `pending` with no run; every minute this starts the\n // oldest such rows (\u2265 30 s old, 20 per tick) with the SAME idempotency key,\n // stops at the first 429 (or at a refusal that means the setup is wrong),\n // and fails \u2014 and tells the owner of \u2014 a row that waited over an hour.\n // overlap 'skip': a slow tick is never doubled by the next one.\n // Free plan: a function cron may fire at most every 15 minutes \u2014 use\n // '*/15 * * * *' there (README \"The backlog\").\n 'redrive-pending': {\n entry: './functions/redrive-pending.ts',\n trigger: { kind: 'cron', schedule: '* * * * *', overlap: 'skip' },\n scopes: ['cms:read', 'cms:write', 'notifications:send'],\n // vxil_jobs_key: an API key of this backend holding ONLY jobs:write \u2014 a\n // cron tick has no signed-in user to hold credits for, so the start is\n // made as your trusted server (README \"The backlog\")\n secrets: ['secret:vxil_jobs_key', 'secret:render_url', 'secret:render_token'],\n egressAllow: [],\n },\n\n // job.generation.completed | failed \u2192 write the failure cause onto the row\n // and tell the owner. A non-2xx is retried; the notification's\n // Idempotency-Key (one per run) makes a redelivery send nothing twice.\n 'notify-ready': {\n entry: './functions/notify-ready.ts',\n trigger: { kind: 'webhook', source: 'job.generation.', retry: { maxAttempts: 3 } },\n scopes: ['cms:read', 'cms:write', 'notifications:send'],\n egressAllow: [],\n },\n },\n\n secrets: {\n render_url: {\n feature: 'functions',\n description: 'your render endpoint \u2014 the https URL vxil POSTs each render to (a Trigger.dev relay, or your own container endpoint)',\n },\n render_token: {\n feature: 'functions',\n description: 'a long random token your render endpoint checks on the Authorization header (Bearer \u2026)',\n },\n vxil_jobs_key: {\n feature: 'functions',\n description: 'an API key of this backend holding ONLY jobs:write \u2014 redrive-pending starts backlogged renders with it (vxil keys mint --name render-redrive --scopes jobs:write). jobs:write also lets it cancel or replay any run, enqueue jobs and manage schedules and flow rules: a server key, kept only here',\n },\n },\n});\n",
|
|
18563
|
-
"readme": "# Render Farm \u2014 long renders on a runtime you rent, with vxil holding the credits, the deadline and the callback\n\n```bash\nmkdir my-renders && cd my-renders\nvxil init --template render-farm # init scaffolds into the CURRENT directory\nprintf '%s' \"$RENDER_URL\" | vxil secrets set functions/render_url # your render endpoint (https)\nprintf '%s' \"$RENDER_TOKEN\" | vxil secrets set functions/render_token # a long random token it checks\n# the backlog re-driver's key: an API key of this backend holding ONLY jobs:write\nvxil login # keys mint needs a dashboard session, not a project key\nvxil keys mint --name render-redrive --scopes jobs:write --json | jq -r .api_key | vxil secrets set functions/vxil_jobs_key\nvxil push\n```\n\n> **Plan note.** The functions deploy on the Free plan when the project's workload is `staging` or\n> `development` (`vxil projects workload <slug> development`, or create it with\n> `vxil projects create <slug> --workload development`). On a Free `production` project, `vxil push` stops before it writes anything, naming the plan and the ways out: change the workload or upgrade to Developer, or run `vxil push --skip-functions` to apply the collections and config without the functions.\n> On the Free plan a function cron may also fire at most every 15 minutes, so change `redrive-pending`'s\n> schedule to `'*/15 * * * *'` there. The blueprint is written for Developer and up, where it runs every\n> minute. Everything else works the same; a backlog just drains more slowly.\n\nA video render, a transcode, a headless-browser capture: minutes of CPU, ffmpeg or Chromium. That does\nnot fit in a vxil function (a delivered trigger gets about a minute), and vxil will not grow a container\ntier or a workflow engine to run it. So the work runs on **a runtime you rent** \u2014 Trigger.dev, Modal,\na container on your own cloud account \u2014 and vxil keeps the four things that must survive while it\nruns:\n\n| vxil holds | so that |\n|---|---|\n| **the credits** reserved for the render | a failed, abandoned or cancelled render gives them back, and a user can never start more than they can pay for |\n| **the deadline** | a render your runtime never reports on fails and refunds after 30 minutes (at most one hour) |\n| **the signed completion callback** | your runtime needs no vxil key: the URL it is handed is the credential for that one render |\n| **the status row** | the app reads (or subscribes to) one `renders` row: `pending \u2192 processing \u2192 completed | failed`, plus everything your runtime sent back |\n\n**What this blueprint teaches that the others do not:** the hand-off to **your own** long-running\nruntime through a **webhook-mode generation run** \u2014 the contract your endpoint and your worker must\nkeep, and two complete runtime options below. (`fal-media` shows the same lane against a vendor queue\nAPI; `job-runner` shows a provider call vxil polls.)\n\n## What you get\n\n- **`renders`** \u2014 one row per render, owned by the user who asked for it (`strictEndUserScope`: a\n signed-in user reads only their own). `render_key` is the owner + `:` + the client's `request_key`\n (the composition is enforced by a `beforeWrite` hook) and is **unique**, so a double tap or a retried\n request finds the first row instead of starting a second render. Your runtime's answer lands on the\n row: `output_url`, `duration_s`, `progress`, `stage`, `message`.\n- **`request-render`** (http function, end-user mode) \u2014 creates-or-finds the row, then starts ONE\n generation run: your endpoint (`render_url`), your token on its `Authorization` header, a status\n mirror onto the row, `reserve_credits` for the render (5 `render_credits`), a 30-minute deadline, and\n the `render_key` as the run's `idempotency_key` \u2014 so a re-driven start gets the same run back. The\n credits it holds are the function's fixed price, never a value read from the row.\n- **`redrive-pending`** (cron function, every minute, `overlap: 'skip'`) \u2014 drains the backlog. When\n the project already has `generation.maxConcurrent` renders in flight, `request-render` answers `429`\n (with `queued: true`) and the row waits `pending` with no run. This function starts those rows,\n oldest first, with the **same** idempotency key, and tells the owner when one never starts\n ([the backlog](#the-backlog-maxconcurrent-429-and-the-re-driver)).\n- **`notify-ready`** (webhook function on `job.generation.`) \u2014 writes the failure cause onto the row (a\n platform class such as `GenerationExpired` gets a short human hint after it) and sends the owner one\n message per run (`Idempotency-Key: render-ready:<run_id>`). A render refused for too few credits\n keeps `insufficient_credits`; when `request-render` refused it, the caller already got the `402`\n and nothing is sent, and when the re-driver started it (the user last heard \"queued\"), the owner\n is told once (`Idempotency-Key: render-not-started:<item_id>`, the re-driver's own key). It\n **branches on `error_class`**, because not every `job.generation.failed` is a render that failed:\n `PaymentsUnavailable` (the credits could not be held at the start \u2014 the row is still queued and a\n new run follows) is skipped (and when a re-sent `request-render` raced that start and linked its\n run, the row is unlinked \u2014 `run_id: null`, only while it is still `pending` on that run \u2014 so\n `redrive-pending` starts it), and `Cancelled` (your own cancel) writes `error: cancelled` and sends\n nothing. Both are also lower-level events (`warn` / `info`), so they stay out of an immediate\n failure digest.\n- **credits** \u2014 a payments integration on the `mock` provider (no provider account needed to try it).\n \"Credits\" are usage units you meter, not money.\n\nThe row is a **view** for the app; the run and the ledger are the truth. If your app's client key\ncarries `cms:write`, a signed-in user can edit their own `renders` row (say, set `status` to\n`completed`) \u2014 that changes nothing they are charged or given. Anything that grants something on\ncompletion should read the run (`GET /v1/jobs/runs/{run_id}`) or react to `job.generation.completed`,\nas `notify-ready` does \u2014 or give the client key only `cms:read`.\n\n## The contract your runtime keeps\n\nWhatever runs the render, these are the only things it has to do:\n\n| Step | What arrives / what to send |\n|---|---|\n| **Start** | vxil `POST`s your `render_url` with `Authorization: Bearer <render_token>`, the header `x-vxil-run-id` and JSON `{ render_id, user_id, composition, props, payload: { generation_id, correlation_id, deadline_at }, callback_url }` (`user_id` is the render's owner). Check the token, **store or queue the work and answer `2xx` within 20 seconds** \u2014 never render inline, and never wait on a container's cold start. When you cannot take the work right now, say so at once with a `503` instead of holding the request open; the start is then sent again later, as after a timeout. `408` / `429` / `5xx` / a timeout is retried with backoff (30 s, then 1, 2 and 4 minutes, doubling up to 10 minutes; a `Retry-After` on your answer sets the wait, 1 s to 10 min, never past the deadline; `max_attempts` counts start calls) until the run's attempts are used \u2014 with the default five, the last try comes 7 to 8 minutes after the first, so leave room for that in the deadline. Any other `4xx` ends the run and refunds the credits. A start can arrive more than once (a lost answer is retried), so **dedupe by the run id** (`x-vxil-run-id`; in this blueprint `render_id` = `payload.generation_id` names the same single run), never by a hash of the content ([why](#one-render-one-run-dedupe-a-doubled-start)). |\n| **Progress** (optional, best-effort) | `POST callback_url` with `{\"status\": \"processing\", \"progress\": 40, \"stage\": \"encoding\"}`. A processing ping moves the row to `processing` and writes its `progress` (0\u2013100) / `stage` (\u2264 64 chars) / `message` (\u2264 200 chars) onto the row \u2014 `request-render` asks for that with `status_mirror.progress_fields` \u2014 so the app shows live progress by watching the row. Other keys on a ping are not written to the row (the run keeps the latest report, `GET /v1/jobs/runs/{run_id}` \u2192 `progress`). **At most one ping every 5 seconds**, sending the latest state; a ping that fails or answers `429` is simply dropped ([callback limits](#callback-limits-progress-is-best-effort-the-final-post-must-arrive)). |\n| **Done** (must arrive) | `POST callback_url` with `{\"status\": \"completed\", \"output_url\": \"https://\u2026\", \"duration_s\": 31.2, \"progress\": 100, \"stage\": \"done\"}` \u2014 or, when the output is uploaded into this project's files, `{\"status\": \"completed\", \"output_file\": \"obj_\u2026\", \"duration_s\": 31.2, \"progress\": 100, \"stage\": \"done\"}` ([below](#keys-stay-out-of-the-container-the-recommended-shape)). At most 256 KiB, and **every key a declared field of `renders`** (add a field before you send a new key; [test it](#a-contract-test-for-your-runtime)). The credits are committed and every key is written onto the row. **Retry this post on `429`, `5xx` and network errors, honouring `Retry-After`, until `deadline_at`** \u2014 never drop it (`postCallback` below). |\n| **Failed** (must arrive) | `POST callback_url` with `{\"status\": \"failed\", \"error\": \"render_failed\", \"hint\": \"ffmpeg exited 1: \u2026\"}`. The credits are refunded; `error` (a short code) and `hint` reach `notify-ready` as `error_class` / `error_hint` and land on the row's `error`. Retried exactly like `completed`. |\n| **Never answers** | the run fails at the deadline (`GenerationExpired`), the credits are refunded, and a callback after that changes nothing. |\n| **The deadline** | `payload.deadline_at` (an ISO time) is when vxil stops waiting \u2014 to the second: any callback that arrives at or after it ends the run as `GenerationExpired` (credits refunded), and a `completed` posted then is answered `{ generation_status: \"failed\", expired: true }` \u2014 the output exists, but the user was refunded and the row says failed. Check it before each attempt starts: a render that cannot finish by then should post `failed` and stop. |\n\n`callback_url` needs no other credential \u2014 and nothing else should see it. A repeated `completed` or\n`failed` post is answered with the settled state and changes nothing, so your worker can safely retry\nits own callback on a network error. The output bytes stay where your runtime wrote them (your bucket,\nyour CDN) and vxil stores the keys, not the file, unless a coordinator uploads the output into this\nproject's files ([next section](#keys-stay-out-of-the-container-the-recommended-shape)).\n\nEach start request also carries an `X-Vxil-Jobs-Signature` header (verifiable with your project's\njobs signing secret, `GET /v1/jobs/signing-secret`) and `x-vxil-run-id`. This blueprint uses the\nbearer token because it is one string comparison in any language.\n\n### Callback limits: progress is best-effort, the final post must arrive\n\nEach render's `callback_url` has its **own** budget at vxil's edge: about 120 posts a minute for that\none run. Every post over it, pings included, is counted against a small separate allowance (about 20\na minute), and on that allowance only a `completed` or `failed` post is accepted. A runtime that pings\nevery 5 seconds never gets near either number. One that sends more than about 140 posts in a minute\nuses the allowance up as well, and its final post waits for the next minute. All of a project's\ncallbacks together also share a ceiling of about 3,000 a minute, plus the same again for final posts.\nEvery number here is best-effort (counted per serving machine). The signature in the URL identifies\nthe run, so fifty renders posting from one shared egress IP do not share a budget. Over a budget the\nanswer is `429 rate_limited` with `Retry-After: 30`.\n\nTwo rules follow, and the helpers below keep both:\n\n- **Progress is best-effort.** Post at most one `processing` ping every 5 seconds, carrying the latest\n state; a ping that fails or answers `429` is dropped, and the next one carries the newer state.\n- **The final post must arrive.** A `completed` or `failed` post that answers `429`, `5xx` or never\n gets an answer is sent again, after `Retry-After` when there is one, until `deadline_at`. A repeat\n of a final post is harmless: it is answered with the settled state and changes nothing. Any other\n `4xx` means the URL or the body is wrong, so retrying will not help.\n\n```ts\n// callbacks.ts \u2014 for the coordinator, a task, or any worker that posts to callback_url\n/** completed / failed: MUST arrive. Retries 429, 5xx and network errors (honouring Retry-After)\n * until `untilMs`; throws only when it cannot deliver in time, or on another 4xx. */\nexport async function postCallback(callbackUrl: string, body: Record<string, unknown>, untilMs: number): Promise<void> {\n for (let attempt = 0; ; attempt++) {\n let res: Response | undefined;\n try {\n res = await fetch(callbackUrl, {\n method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body),\n });\n } catch { /* a network error: send it again */ }\n if (res && res.ok) return;\n if (res && res.status !== 429 && res.status < 500) throw new Error(`callback refused: ${res.status}`);\n const retryAfterS = Number(res?.headers.get('retry-after'));\n const waitMs = retryAfterS > 0 ? retryAfterS * 1000 : Math.min(1000 * 2 ** attempt, 30_000);\n if (Date.now() + waitMs > untilMs) throw new Error(`callback not delivered in time (last answer: ${res?.status ?? 'none'})`);\n await new Promise((r) => setTimeout(r, waitMs));\n }\n}\n\n/** processing: best-effort, at most one post every `gapMs`; a refused or failed ping is dropped. */\nexport function progressReporter(callbackUrl: string, gapMs = 5_000) {\n let last = 0;\n return async (p: { progress?: number; stage?: string; message?: string }): Promise<void> => {\n if (Date.now() - last < gapMs) return; // coalesced: the next ping carries newer state\n last = Date.now();\n await fetch(callbackUrl, {\n method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify({ status: 'processing', ...p }),\n }).catch(() => undefined);\n };\n}\n```\n\n### One render, one run: dedupe a doubled start\n\nvxil sends the start again when it did not get your answer: a timeout, a dropped connection, a `5xx`.\nSo the same start can reach you twice, even while the first copy is still being handled. Claim the\nwork under the **run id** (`x-vxil-run-id`) before you launch anything. Answer `202` to a repeat\nonly once the work is launched. While the first copy is still launching, answer `503`, so the start\nstays open and vxil sends it again; a `202` would end vxil's retries even if that first launch then failed. In this blueprint `render_id` (= `payload.generation_id`, the row's id)\nnames the same single run, because `request-render` starts one run per row.\n\nNever key the claim by a hash of the content (the composition and props). Two users who ask for the\nsame render produce two runs with one hash: the second start would find the first's claim and be\nrefused, or overwrite the first's `callback_url`, and that run would wait out its deadline and refund.\nIf you want identical renders to share their output, claim by run id and look the output up by hash\nas a separate step.\n\n## Keys stay out of the container (the recommended shape)\n\nThe render container is the part of your system that runs the most third-party code (ffmpeg,\nChromium, fonts and media from the user's props), so give it **no vxil key at all**. This is the\nshape the blueprint recommends, whatever runs the container:\n\n```\nvxil \u2500\u2500start\u2500\u2500\u25B6 coordinator \u2500\u2500launch\u2500\u2500\u25B6 container\n (render_url) \u2502 progress / failed \u2500\u2500\u2500\u2500\u2500\u2500\u25B6 callback_url (keyless)\n \u25B2 output bytes \u2500\u2500\u2500\u2500\u2500\u2518\n \u2502\n \u2514\u2500 mints the upload URL with ITS files:write key, PUTs the bytes,\n completes the object, POSTs \"completed\" + output_file \u2500\u2500\u25B6 callback_url\n```\n\n- **The files feature is on.** The blueprint's config enables it (`files: { enabled: true }`) and\n declares `output_file` as a `file` field; without it every upload-url call is refused and the\n render waits out its deadline. Its per-object ceiling, `maxObjectBytes`, is 100 MB by default,\n enough for the buffered coordinator below; raise it for long or high-bitrate renders.\n- **The coordinator** is your `render_url`: a small endpoint in your own account \u2014 an edge worker\n with a per-render lock, or a route on any thin server. It is the only piece that holds a vxil key,\n and that key holds only **`files:write`** (`vxil keys mint --name render-uploads --scopes files:write`).\n- **The container** gets the job, the keyless `callback_url` (for `processing` pings and for\n `failed`), an `output_url` on the coordinator and an **output ticket** for it: a token signed for\n that one render and useless after its deadline, sent on the `Authorization` header (never in the\n URL, where access logs would keep it).\n- **The upload happens once the size is known.** The container POSTs the finished file to its\n ticket. The coordinator reads it, mints the files upload URL for exactly that size\n (the quota pre-check uses it), PUTs the bytes, completes the object, and only then posts\n `completed` with the object id as `output_file`. A crash anywhere before that leaves the render\n open, and it refunds at the deadline like any other.\n\n```ts\n// coordinator.ts \u2014 your render_url. A standard fetch handler (an edge worker, or a Node 18+ adapter).\n// Env: RENDER_TOKEN (= the vxil secret render_token), TICKET_SECRET (a long random string),\n// VXIL_FILES_KEY (an API key holding ONLY files:write), VXIL_BASE (https://api.vxil.com).\nimport { postCallback } from './callbacks';\n\ntype Start = {\n render_id: string; user_id: string; composition: string; props?: Record<string, unknown>;\n payload: { generation_id: string; deadline_at: string }; callback_url: string;\n};\ntype Ticket = { run_id: string; render_id: string; user_id: string; callback_url: string; exp: number };\ntype Env = { RENDER_TOKEN: string; TICKET_SECRET: string; VXIL_FILES_KEY: string; VXIL_BASE: string };\n\nconst enc = new TextEncoder();\nconst b64u = (b: ArrayBuffer | Uint8Array) =>\n btoa(String.fromCharCode(...new Uint8Array(b))).replace(/\\+/g, '-').replace(/\\//g, '_').replace(/=+$/, '');\nconst unb64u = (s: string) => Uint8Array.from(atob(s.replace(/-/g, '+').replace(/_/g, '/')), (c) => c.charCodeAt(0));\nconst hmacKey = (secret: string) =>\n crypto.subtle.importKey('raw', enc.encode(secret), { name: 'HMAC', hash: 'SHA-256' }, false, ['sign', 'verify']);\nasync function sealTicket(env: Env, t: Ticket): Promise<string> {\n const body = b64u(enc.encode(JSON.stringify(t)));\n return `${body}.${b64u(await crypto.subtle.sign('HMAC', await hmacKey(env.TICKET_SECRET), enc.encode(body)))}`;\n}\nasync function openTicket(env: Env, raw: string): Promise<Ticket | null> {\n const [body, sig] = raw.split('.');\n if (!body || !sig) return null;\n try {\n // crypto.subtle.verify compares in constant time (a `!==` on the signature would not)\n if (!(await crypto.subtle.verify('HMAC', await hmacKey(env.TICKET_SECRET), unb64u(sig), enc.encode(body)))) return null;\n const t = JSON.parse(new TextDecoder().decode(unb64u(body))) as Ticket;\n return t.exp > Date.now() ? t : null;\n } catch {\n return null; // not base64url / not JSON\n }\n}\n/** The output's file extension, from the Content-Type the container sends. */\nconst EXT: Record<string, string> = {\n 'video/mp4': 'mp4', 'video/webm': 'webm', 'image/gif': 'gif', 'image/png': 'png', 'image/jpeg': 'jpg', 'application/pdf': 'pdf',\n};\n/** Sent with a 503: when to send the request again. */\nconst AGAIN_IN_30S = { 'retry-after': '30' };\nconst vxil = (env: Env, path: string, body?: unknown) => fetch(`${env.VXIL_BASE}${path}`, {\n method: 'POST',\n headers: { authorization: `Bearer ${env.VXIL_FILES_KEY}`, 'content-type': 'application/json' },\n ...(body ? { body: JSON.stringify(body) } : {}),\n});\n\nexport default {\n async fetch(req: Request, env: Env): Promise<Response> {\n const url = new URL(req.url);\n\n // 1. the start: check the token, launch the container, answer inside 20 s\n if (req.method === 'POST' && url.pathname === '/start') {\n if (req.headers.get('authorization') !== `Bearer ${env.RENDER_TOKEN}`) return new Response('unauthorized', { status: 401 });\n const s = (await req.json()) as Start;\n // a run started before request-render sent user_id (an older copy of this\n // blueprint): a server-mode upload must name its user, so refuse the start \u2014\n // a 4xx ends that run and refunds it, instead of a 422 at upload time\n if (!s.user_id) return new Response('start body has no user_id: redeploy request-render', { status: 400 });\n // one run, one render: claim the work under the RUN id before launching anything. The claim\n // says `launching` (and expires after a minute) until the launch succeeds, then `launched`.\n // A start vxil re-sends (a lost answer) is acknowledged with 202 only once the work is\n // `launched`; while another copy is still launching it gets 503, so vxil keeps retrying\n // instead of treating a launch that may yet fail as accepted.\n // Never claim by a content hash: two users' identical renders are two runs.\n const runId = req.headers.get('x-vxil-run-id') ?? s.payload.generation_id;\n const claimKey = `start:${runId}`;\n if (!(await store.claim(claimKey, 'launching', 60_000))) {\n if ((await store.get(claimKey)) === 'launched') return Response.json({ accepted: true, duplicate: true }, { status: 202 });\n return new Response('this render is still being launched', { status: 503, headers: AGAIN_IN_30S });\n }\n const ticket = await sealTicket(env, {\n run_id: runId, render_id: s.render_id, user_id: s.user_id, callback_url: s.callback_url,\n exp: Date.parse(s.payload.deadline_at), // useless once vxil stops waiting\n });\n try {\n await launchContainer({ // YOUR container platform's API: it must QUEUE\n run_id: runId, render_id: s.render_id, // the job and return at once \u2014 never wait.\n // Pass run_id as its idempotency key if it has one\n composition: s.composition, props: s.props ?? {},// here for a cold start\n deadline_at: s.payload.deadline_at,\n callback_url: s.callback_url, // for processing pings and `failed`\n output_url: `${url.origin}/output`, // POST the file here, with\n output_ticket: ticket, // Authorization: Bearer <output_ticket>\n });\n } catch {\n await store.release(claimKey); // not launched: let vxil's retry try again\n return new Response('cannot take the render right now', { status: 503, headers: AGAIN_IN_30S });\n }\n await store.put(claimKey, 'launched'); // no expiry: every later copy is a duplicate\n return Response.json({ accepted: true }, { status: 202 });\n }\n\n // 2. the output: the container POSTs the finished file here (again, on any non-2xx answer)\n if (req.method === 'POST' && url.pathname === '/output') {\n const t = await openTicket(env, (req.headers.get('authorization') ?? '').replace(/^Bearer /, ''));\n if (!t) return new Response('bad or expired ticket', { status: 403 });\n const contentType = (req.headers.get('content-type') ?? '').split(';')[0]!.trim().toLowerCase();\n const ext = EXT[contentType];\n if (!ext) return new Response(`send the output's Content-Type (one of: ${Object.keys(EXT).join(', ')})`, { status: 415 });\n // a re-sent output after the upload already happened: skip straight to the settle\n let object_id = await store.get(`output:${t.run_id}`);\n if (!object_id) {\n // buffered: fine for outputs of tens of MB \u2014 see \"Very large outputs\" below\n const bytes = await req.arrayBuffer();\n const size = bytes.byteLength;\n if (size === 0) return new Response('empty output', { status: 400 });\n\n const minted = await vxil(env, '/v1/files/upload-url', {\n user_id: t.user_id, filename: `${t.render_id}.${ext}`, content_type: contentType, size_bytes: size,\n });\n if (!minted.ok) return new Response(`upload-url ${minted.status}`, { status: 502 }); // the container sends the file again\n const m = ((await minted.json()) as { data: { object_id: string; upload_url: string } }).data;\n const put = await fetch(m.upload_url, { method: 'PUT', headers: { 'content-type': contentType }, body: bytes });\n if (!put.ok) return new Response(`upload ${put.status}`, { status: 502 });\n const done = await vxil(env, `/v1/files/${encodeURIComponent(m.object_id)}/complete`);\n if (!done.ok) return new Response(`complete ${done.status}`, { status: 502 });\n object_id = m.object_id;\n await store.put(`output:${t.run_id}`, object_id);\n }\n\n // 3. settle the render on the keyless callback: a MUST-ARRIVE post. Retry here for up to a\n // minute (never past the deadline); if it still did not land, answer 503 and the container\n // sends the output again \u2014 the note above turns that into one more settle attempt.\n try {\n await postCallback(t.callback_url,\n { status: 'completed', output_file: object_id, progress: 100, stage: 'done' },\n Math.min(t.exp, Date.now() + 60_000));\n } catch (e) {\n return new Response(`callback: ${String(e)}`, { status: 503, headers: AGAIN_IN_30S });\n }\n return Response.json({ object_id });\n }\n return new Response('not found', { status: 404 });\n },\n};\n\ndeclare function launchContainer(job: Record<string, unknown>): Promise<void>; // your container platform's API\n/** Your coordinator's own small store: a key-value namespace, a table, a per-key lock. `claim` is\n * an insert-if-absent (of `value`, expiring after `ttlMs`) that answers true for the FIRST caller\n * only; `put` writes without an expiry. */\ndeclare const store: {\n claim(key: string, value: string, ttlMs: number): Promise<boolean>; release(key: string): Promise<void>;\n get(key: string): Promise<string | null>; put(key: string, value: string): Promise<void>;\n};\n```\n\nThe container's side is three kinds of HTTP call and no vxil key: `processing` pings to\n`callback_url`, the file to `output_url` (with `Authorization: Bearer <output_ticket>` and the\nfile's `Content-Type`), and `{\"status\": \"failed\", \u2026}` to `callback_url` if it gives up. The\ncontainer sends its pings through `progressReporter` and its `failed` through `postCallback`\n([callbacks.ts](#callback-limits-progress-is-best-effort-the-final-post-must-arrive)), and it sends\nthe output again on any non-`2xx` answer until `deadline_at`. Five notes on the shape:\n\n- **A doubled start launches one container.** The coordinator claims `start:<run id>` as\n `launching` before it launches and marks it `launched` once the launch succeeded. A start vxil\n re-sends is answered `202` and launches nothing only when the claim says `launched`; while the\n first copy is still launching, the re-sent one is answered `503`, so vxil keeps the start alive.\n (A `202` there would end vxil's retries, and if the first launch then failed nothing would run\n and the render would refund at its deadline.) When the launch fails, the coordinator releases\n the claim and answers `503` at once, and vxil sends the start again later. A coordinator that\n dies mid-launch leaves a `launching` claim that expires after a minute, so a later copy launches;\n pass the run id as your container platform's idempotency key, where it has one, so that copy\n cannot start a second container if the first launch did go through.\n- **A retried output post** (the container saw a network error after the coordinator had uploaded)\n finds `output:<run id>` in the coordinator's store, skips the upload and only sends the\n `completed` post again. A repeated `completed` is answered with the settled state and changes\n nothing, so the final post can be retried as often as it takes.\n- **Very large outputs.** The coordinator above holds the whole file in memory, and passing gigabytes\n through it costs its bandwidth too. For big files, have the container report the size first\n (`POST /output-url` with its ticket on the `Authorization` header and `{ size_bytes }`) and let the coordinator answer with the\n presigned upload URL it minted; the container PUTs straight to it, and the coordinator completes\n the object and posts `completed` when the container says it is done. The container still holds no\n key: a presigned URL is good for one object for a few minutes.\n- **Upgrading an earlier copy of this blueprint.** Renders started before `request-render` put\n `user_id` in the start body have none, and a server-mode upload must name its user. The\n coordinator refuses such a start with a `400`, which ends that run and refunds it at once;\n redeploy `request-render` (`vxil push`) before you point `render_url` at the coordinator.\n- **Serving it.** The row now holds a files object id. Read it back with a signed download URL, or\n publish it from a settle function with a server key (`vx.files.publish(object_id)`) for a stable\n public URL served from the edge cache.\n\nOptions A and B below show the same contract with the runtime posting `completed` itself (an\n`output_url` in your own bucket). Either can adopt the coordinator: point `render_url` at it, and have\nstep 1 trigger the Trigger.dev task or spawn the Modal function.\n\n## Option A \u2014 Trigger.dev (v4)\n\nTrigger.dev runs the render as a task on a machine you pick, with ffmpeg or Chromium baked into the\nimage, retries, and its own run dashboard. Two pieces: a **relay endpoint** that turns vxil's start\nrequest into a Trigger.dev trigger, and the **task**.\n\n**Why a relay, and not `render_url` pointed straight at Trigger.dev's trigger API?** vxil sends\n`callback_url` beside `payload` at the top level of the body, and Trigger.dev's trigger API passes\nonly `payload` to the task \u2014 the task would never see where to report. The relay is ~30 lines and is\nalso where your `render_token` is checked. Host it anywhere that serves https: a serverless function on\nyour web host, a small edge worker, a route in your existing API.\n\n```ts\n// relay.ts \u2014 your render_url. Standard fetch handler (edge worker / serverless function / Node 18+ adapter).\n// Env: RENDER_TOKEN (the same value as the vxil secret render_token), TRIGGER_SECRET_KEY (tr_prod_\u2026 / tr_dev_\u2026).\ntype Start = {\n render_id: string; composition: string; props?: Record<string, unknown>;\n payload: { generation_id: string; correlation_id?: string; deadline_at: string }; callback_url: string;\n};\n\nexport default {\n async fetch(req: Request, env: { RENDER_TOKEN: string; TRIGGER_SECRET_KEY: string }): Promise<Response> {\n if (req.method !== 'POST') return new Response('method not allowed', { status: 405 });\n if (req.headers.get('authorization') !== `Bearer ${env.RENDER_TOKEN}`) {\n return new Response('unauthorized', { status: 401 }); // a 4xx ends the vxil run (and refunds)\n }\n const s = (await req.json()) as Start;\n const res = await fetch('https://api.trigger.dev/api/v1/tasks/render-video/trigger', {\n method: 'POST',\n headers: { authorization: `Bearer ${env.TRIGGER_SECRET_KEY}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n payload: {\n render_id: s.render_id, composition: s.composition, props: s.props ?? {},\n callback_url: s.callback_url, deadline_at: s.payload.deadline_at,\n },\n options: {\n // a start vxil re-sends (a lost answer) triggers the SAME Trigger.dev run\n idempotencyKey: `render:${s.render_id}`,\n // a render still queued after 3 minutes is dropped (it never runs, and no\n // onFailure fires \u2014 vxil refunds it at its deadline). Part of the budget below.\n ttl: '3m',\n tags: [`render_${s.render_id}`],\n },\n }),\n });\n if (res.ok) return Response.json({ accepted: true }, { status: 202 });\n // Trigger.dev busy or down: answer 503 and vxil tries the start again; anything else ends the run\n return new Response(`trigger.dev ${res.status}`, { status: res.status === 429 || res.status >= 500 ? 503 : 400 });\n },\n};\n```\n\n```ts\n// trigger.config.ts \u2014 ffmpeg and Chromium in the task image\nimport { defineConfig } from '@trigger.dev/sdk';\nimport { ffmpeg } from '@trigger.dev/build/extensions/core';\nimport { puppeteer } from '@trigger.dev/build/extensions/puppeteer';\n\nexport default defineConfig({\n project: '<your project ref>',\n dirs: ['./trigger'],\n maxDuration: 600, // CPU seconds PER ATTEMPT \u2014 the task sets its own; see the budget below\n build: { extensions: [ffmpeg(), puppeteer()] }, // puppeteer also needs PUPPETEER_EXECUTABLE_PATH set in the Trigger.dev env\n});\n```\n\n```ts\n// trigger/render-video.ts \u2014 the task: render, upload, report to vxil\nimport { task, metadata, logger } from '@trigger.dev/sdk';\nimport { postCallback, progressReporter } from '../callbacks'; // the helpers above\n\ntype Payload = {\n render_id: string; composition: string; props: Record<string, unknown>;\n callback_url: string; deadline_at: string; // when vxil stops waiting (ISO)\n};\n\n/** The longest one attempt takes, wall clock, with margin. An attempt that cannot\n * finish before deadline_at does not start: it tells vxil, so the credits come back now. */\nconst ATTEMPT_WALL_MS = 12 * 60_000;\n\n/** completed / failed: retried on 429 / 5xx / network errors (Retry-After honoured) until the\n * deadline \u2014 a repeated final post is a no-op on vxil's side, so retrying is always safe. */\nconst report = (p: Payload, body: Record<string, unknown>) =>\n postCallback(p.callback_url, body, Date.parse(p.deadline_at));\n\nexport const renderVideo = task({\n id: 'render-video',\n machine: 'large-1x', // 4 vCPU / 8 GB \u2014 size to your renders\n maxDuration: 600, // CPU seconds per attempt (10 min) \u2014 not wall time\n retry: { maxAttempts: 2, minTimeoutInMs: 5_000, maxTimeoutInMs: 30_000 },\n run: async (p: Payload) => {\n if (Date.now() + ATTEMPT_WALL_MS > Date.parse(p.deadline_at)) {\n // too late to finish inside vxil's deadline: refund now, and do not retry\n await report(p, { status: 'failed', error: 'deadline', hint: 'no time left for another attempt' });\n return { skipped: 'deadline' };\n }\n metadata.set('stage', 'rendering'); // Trigger.dev's own run view\n const progress = progressReporter(p.callback_url); // best-effort, at most one ping per 5 s\n await progress({ stage: 'rendering', progress: 0 });\n\n // \u2026your render: drive Chromium for frames, run ffmpeg \u2014 call progress({ progress, stage })\n // as often as you like (it coalesces) \u2014 and write the file\n // to YOUR bucket keyed by render_id (so a retried attempt overwrites, not duplicates)\u2026\n const outputUrl = `https://cdn.example.com/renders/${p.render_id}.mp4`;\n const durationS = 31.2;\n logger.info('rendered', { render_id: p.render_id, outputUrl });\n if (Date.now() > Date.parse(p.deadline_at)) {\n // vxil has already failed and refunded this render: the post below is answered\n // with that settled state. Your sizing is off \u2014 widen the budget below.\n logger.warn('finished after the vxil deadline', { render_id: p.render_id });\n }\n\n await report(p, {\n status: 'completed', output_url: outputUrl, duration_s: durationS, progress: 100, stage: 'done',\n });\n return { output_url: outputUrl };\n },\n // after the last attempt THROWS: tell vxil, so the credits come back now, not at the\n // deadline. Not called when an attempt exceeds maxDuration or the run expires on its\n // ttl \u2014 those refund only at vxil's deadline.\n onFailure: async ({ payload, error }) => {\n await report(payload, {\n status: 'failed', error: 'render_failed', hint: String(error instanceof Error ? error.message : error).slice(0, 200),\n });\n },\n});\n```\n\n**Budget the wall clock.** Trigger.dev's limits and vxil's deadline are separate clocks, and only\nvxil's refunds. Size them so a render always ends \u2014 `completed` or `failed` \u2014 before vxil's deadline:\n\n```\nttl + maxAttempts \xD7 (longest attempt, wall clock) + retry backoff < timeout.after_ms\n3 min + 2 \xD7 12 min + \u2264 1 min = 28 min < 30 min\n```\n\n`maxDuration` counts **CPU time per attempt**, not wall time across the run, so it does not bound\nthe sum: the `deadline_at` check at the start of each attempt does. If your renders need more, raise\n`RENDER_DEADLINE_MS` in `request-render` (up to `generation.maxTimeoutMs`, one hour) and resize the\nrest to fit.\n\n**Be honest with yourself about three things before you ship on Trigger.dev Cloud:**\n\n- **Data residency.** Trigger.dev Cloud keeps its operational and log data \u2014 including each run's\n payload \u2014 in the US (us-east-1), even when the machines run elsewhere. Your render props and the\n `callback_url` pass through it. If that rules it out, self-host Trigger.dev for your own app, or use\n option B.\n- **The callback URL is a credential for one render.** It appears in Trigger.dev's run payload and\n dashboard. It can settle only that render, and stops mattering once the render is settled.\n- **Two clocks, and `onFailure` is not a guarantee.** Trigger.dev calls `onFailure` only after the last\n attempt throws. A run that exceeds `maxDuration`, or expires on its `ttl` before it starts, ends\n without it \u2014 vxil refunds those at its deadline, not sooner. Keep the budget above, and keep the\n `deadline_at` check, so a late attempt refunds early instead of finishing after the refund.\n\n## Option B \u2014 your own container runtime (Modal, Fly, a container on your own cloud account)\n\nSame contract, no relay: the endpoint you deploy **is** `render_url`. It must answer within 20 seconds,\nso it only checks the token, hands the job to a background worker and answers `2xx`; the worker renders\nand posts back. On Modal:\n\n```python\n# render_app.py \u2014 `modal deploy render_app.py`; render_url = the endpoint's https URL\nimport http.client, json, os, time, urllib.error, urllib.request\nfrom datetime import datetime, timedelta, timezone\nimport modal\nfrom fastapi import HTTPException, Request # also `pip install fastapi` where you run `modal deploy`\n\nimage = (modal.Image.debian_slim()\n .apt_install(\"ffmpeg\", \"chromium\")\n .pip_install(\"fastapi[standard]\"))\napp = modal.App(\"render-farm\", image=image)\nsecrets = [modal.Secret.from_name(\"render-farm\")] # RENDER_TOKEN\n\ndef _post(callback_url: str, body: dict) -> None:\n req = urllib.request.Request(callback_url, data=json.dumps(body).encode(),\n headers={\"content-type\": \"application/json\"}, method=\"POST\")\n with urllib.request.urlopen(req, timeout=30) as res:\n res.read()\n\ndef report(callback_url: str, body: dict, deadline: datetime) -> None:\n \"\"\"completed / failed: MUST arrive. Retries 429, 5xx and network errors (honouring\n Retry-After) until the deadline; a repeated final post changes nothing on vxil's side.\"\"\"\n attempt = 0\n while True:\n wait = min(2 ** attempt, 30)\n attempt += 1\n try:\n _post(callback_url, body)\n return\n except urllib.error.HTTPError as e:\n if e.code != 429 and e.code < 500:\n raise # another 4xx: the URL or the body is wrong\n ra = e.headers.get(\"retry-after\")\n wait = int(ra) if ra and ra.isdigit() else wait\n except (OSError, http.client.HTTPException):\n pass # no answer, or the connection dropped mid-answer\n # (URLError, a reset, RemoteDisconnected, IncompleteRead,\n # a timeout): send it again\n if datetime.now(timezone.utc) + timedelta(seconds=wait) > deadline:\n raise RuntimeError(\"callback not delivered before the deadline\")\n time.sleep(wait)\n\n_last_ping = {}\ndef ping(callback_url: str, body: dict) -> None:\n \"\"\"processing: best-effort, at most one every 5 s; a refused or failed ping is dropped.\"\"\"\n now = time.monotonic()\n if now - _last_ping.get(callback_url, 0.0) < 5:\n return\n _last_ping[callback_url] = now\n try:\n _post(callback_url, {\"status\": \"processing\", **body})\n except Exception:\n pass\n\nATTEMPT_WALL_S = 25 * 60 # = the timeout below; a call cut off there may never reach its except\n\n@app.function(cpu=4, memory=8192, timeout=ATTEMPT_WALL_S, secrets=secrets)\ndef render(job: dict) -> None:\n cb = job[\"callback_url\"]\n deadline = datetime.fromisoformat(job[\"payload\"][\"deadline_at\"].replace(\"Z\", \"+00:00\"))\n if datetime.now(timezone.utc) + timedelta(seconds=ATTEMPT_WALL_S) > deadline:\n # queued too long to finish before vxil stops waiting: refund now\n report(cb, {\"status\": \"failed\", \"error\": \"deadline\", \"hint\": \"started too late to finish\"}, deadline)\n return\n try:\n ping(cb, {\"stage\": \"rendering\", \"progress\": 0})\n # \u2026render with ffmpeg / chromium (ping(cb, {...}) as often as you like),\n # upload to YOUR bucket keyed by job[\"render_id\"]\u2026\n output_url = f\"https://cdn.example.com/renders/{job['render_id']}.mp4\"\n except Exception as e: # the RENDER failed: tell vxil now, so the credits come back before the deadline\n report(cb, {\"status\": \"failed\", \"error\": \"render_failed\", \"hint\": str(e)[:200]}, deadline)\n raise\n # the output exists: only `completed` may follow. Outside the try on purpose, so a\n # trouble delivering it can never turn into a `failed` post that refunds a finished render.\n report(cb, {\"status\": \"completed\", \"output_url\": output_url, \"duration_s\": 31.2,\n \"progress\": 100, \"stage\": \"done\"}, deadline)\n\n@app.function(secrets=secrets)\n@modal.fastapi_endpoint(method=\"POST\")\nasync def start(request: Request):\n if request.headers.get(\"authorization\") != f\"Bearer {os.environ['RENDER_TOKEN']}\":\n raise HTTPException(status_code=401, detail=\"unauthorized\") # a 4xx ends the vxil run\n job = await request.json()\n await render.spawn.aio(job) # queued; returns at once, well inside the 20-second window\n return {\"accepted\": True}\n```\n\n`spawn` queues the call and returns immediately, so the endpoint answers in well under a second. A\nstart can arrive twice (a lost answer is retried), so dedupe on the run id\n(`request.headers[\"x-vxil-run-id\"]`; never a hash of the props): keep the run ids you have spawned in\na `modal.Dict`, or make the render overwrite the same output key.\n\nAnything else that can (1) answer an https POST in under 20 s, (2) run the work in the background and\n(3) POST JSON to a URL fits the same contract: a Fly Machine started per job, a container service with\na queue in front, your own GPU box. vxil does not care what runs the render \u2014 only that the start is\nacknowledged quickly and the callback eventually comes.\n\n## Run it\n\nGive a user some credits from your server (or sell `render_pack_100` through your payments provider):\n\n```bash\ncurl -s -X POST \"https://api.vxil.com/v1/payments/credits/grant\" \\\n -H \"authorization: Bearer $KEY\" -H 'content-type: application/json' \\\n -H 'idempotency-key: welcome-u1' \\\n -d '{\"user_id\":\"<the user id>\",\"credit_type\":\"render_credits\",\"amount\":25,\"source\":\"welcome\"}'\n```\n\nStart a render **with the user's session** (end-user mode \u2014 the held credits are forced onto that\nuser):\n\n```ts\nimport { Vxil } from '@vxil/sdk';\n\n// after `vxil gen`, vx.fn['request-render'] is typed from the function's declared signature\nconst vx = new Vxil({ apiKey: process.env.VXIL_PUBLISHABLE_KEY!, endUserToken: process.env.USER_SESSION! });\nconst started = await vx.fn['request-render']({\n composition: 'promo-30s', request_key: 'promo-1', props: { headline: 'Spring sale' },\n});\n// \u2192 { item_id, run_id, credits: 5 }\n// (or { duplicate: true, request_key, item_id, run_id, status } on a retry)\n```\n\nThen read the row \u2014 or subscribe to its changes \u2014 until `status` is `completed`:\n\n```ts\nif ('item_id' in started) {\n const row = await vx.from('renders').get(started.item_id);\n // row.status \u2192 'completed', row.output_url \u2192 'https://cdn.example.com/renders/\u2026.mp4'\n}\n```\n\n**Try it before you have a runtime.** Point `render_url` at any https endpoint that answers `2xx`\n(a request-bin works) and play the runtime yourself: copy `callback_url` from the request it received,\nthen\n\n```bash\ncurl -s -X POST \"$CALLBACK_URL\" -H 'content-type: application/json' \\\n -d '{\"status\":\"completed\",\"output_url\":\"https://cdn.example.com/x.mp4\",\"duration_s\":12.5,\"progress\":100,\"stage\":\"done\"}'\n# \u2192 { \"data\": { \"run_id\": \"run_\u2026\", \"generation_status\": \"completed\" } } \u2014 and the row says so\n```\n\n## How it fails, and what the user sees\n\n| what happened | the run | the row | the credits |\n|---|---|---|---|\n| runtime posted `completed` | `completed` | `status: completed` + every key it sent | committed |\n| runtime posted `failed` | `failed` | `status: failed`, `error` = its code + hint (written by `notify-ready`) | refunded |\n| runtime never called back | `failed` (`GenerationExpired`) at the deadline | `failed`, `error: GenerationExpired: the render did not finish before its deadline` | refunded |\n| runtime finished after the deadline | `failed` (`GenerationExpired`) \u2014 exact to the second; the late `completed` is answered `failed` / `expired: true` | `failed` | refunded (your compute was spent \u2014 budget the clocks) |\n| the final post was answered `429` (or `5xx`, or got no answer) | still open: nothing is settled until a post lands | unchanged | still held \u2014 `postCallback` sends it again after `Retry-After`; give up only at `deadline_at`, when the run refunds anyway |\n| a progress ping was answered `429` | unchanged | the previous progress stays | unchanged \u2014 drop the ping; the next one carries the newer state |\n| your endpoint answered `5xx` / timed out | start retried with backoff; terminal after the attempts | `processing` \u2192 `failed`, `error: RetryableHttp: the render endpoint kept failing to accept the render (retries exhausted)` (or `NetworkError: \u2026` when it could not be reached) | held until then, then refunded |\n| your endpoint answered another `4xx` (a bad token) | `failed` at once | `failed` | refunded |\n| the user had too few credits (at `request-render`) | ended at once (`ReserveInsufficient`), never started | `failed`, `error: insufficient_credits`; that `request_key` is spent; the caller got the `402`, no message | nothing held |\n| too many renders in flight, payments briefly unreachable, or a jobs-side fault | not created yet (the caller gets `429` / `503` with `queued: true` and `retry_after`, or `502` with `queued: true`) | `pending`, no run \u2014 **queued**: `redrive-pending` starts it when a slot frees up (or the app calls again with the **same** `request_key`) | held when it starts |\n| the user had too few credits when the re-driver started it | ended at once (`ReserveInsufficient`) | `failed`, `error: insufficient_credits`; the owner is told once (they last heard \"queued\") | nothing held |\n| still no free slot an hour later | never created | `failed`, `error: not_started: the render waited too long for a free slot` (written by `redrive-pending`); the owner is told once | nothing held |\n| the re-driver's start was refused (`400` / `401` / `403` / `422`: a non-https or private `render_url`, a wrong or revoked `vxil_jobs_key`) | not created | still `pending` \u2014 the tick stops and reports it (`stopped: { status, code }` in the function's logs); fix the setup and the next tick carries on; the one-hour bound still applies | nothing held |\n\n## The backlog: `maxConcurrent`, `429` and the re-driver\n\n`generation.maxConcurrent` is how many generation runs this project may have **open** at once: 20 by\ndefault, settable up to 200 in the jobs config. This blueprint sets 20; raise it to what your plan\nand your runtime can carry:\n\n```ts\njobs: { enabled: true, generation: { maxConcurrent: 50 /* 1\u2013200, default 20 */ } },\n```\n\nAt the cap a new start is refused with `429` and `Retry-After: 5`. **No run is created and nothing\nis held yet.** The same holds when the credits cannot be held because payments is briefly\nunreachable: the start is refused with `503 payments_unavailable` (nothing started, nothing held;\nit names a 15-second wait), `request-render` answers `503` with `queued: true`, and the re-driver starts\nthe row on a later tick with the same key (a fresh run). Never a free render: that only happens if\nyou opt in with `reserve_credits.on_unavailable: 'proceed'`. `request-render` passes the `429` on to the app with `queued: true` (and the\n`item_id` and `request_key`), and leaves the row `pending` with no `run_id`. That render is\n**queued, not refused**: the re-driver will start it, and hold its credits then, up to an hour\nlater. So the app must treat a `429` with `queued: true` as \"queued\" \u2014 show it, watch the row \u2014 and\nmust **never retry it with a new `request_key`**: that is a second render, and both are charged. A\nretry with the **same** `request_key` is always safe. Rows like that are the backlog, and two\nthings drain it:\n\n1. **`redrive-pending`, every minute.** It reads the oldest `pending` rows with no run that are at\n least 30 s old (`{ status: 'pending', created_at: { $lt: \u2026 }, run_id: null }`, sorted by\n `created_at`, 20 per tick; `status` and `created_at` are index slots, so the read stays cheap at\n any size) and starts each one with the same descriptor and the **same idempotency key** as\n `request-render`, so a row the app is re-driving at the same moment still gets one run. It\n **stops at the first `429`** (or `503 payments_unavailable`), since the rest of the batch would\n get the same answer, and the next tick carries on. Each try is recorded on the row (`redrive_attempts`, `redriven_at`). A `402`\n fails the row with `insufficient_credits`. A `400`, `401`, `403` or `422` also **stops** the tick\n and fails nothing: every re-driven row sends the same descriptor, so a refusal means the setup is\n wrong (a non-https or private `render_url`, a revoked key), not the row; the tick reports it as\n `stopped: { status, code }`. The only thing that fails a waiting row is age: a row still waiting\n after **one hour** is failed with `not_started: the render waited too long for a free slot`.\n Nothing was ever held, so nothing is refunded. In both the `402` and the one-hour case the owner\n is told once (`Idempotency-Key: render-not-started:<item_id>`), because the last thing they heard\n was \"queued\". (The function's scopes include `notifications:send` for that.)\n2. **The app**, calling `request-render` again with the same `request_key`, if it wants the render\n started sooner than the next tick.\n\n`overlap: 'skip'` keeps a slow tick from being doubled by the next one. On the Free plan, run it\nevery 15 minutes (see the plan note at the top).\n\n**Why it has its own key.** A credit hold placed by a function must name the signed-in user the\nfunction acts for, and a cron tick has none. So `redrive-pending` makes the start with\n`vxil_jobs_key`, an API key holding only `jobs:write`, the way your trusted server would. That is\nmore than \"hold credits and start runs\": `jobs:write` also lets the key cancel or replay any run of\nthis project, enqueue any job, and create or change schedules and flow rules, and it can hold\ncredits against any of your users and start runs against any public https endpoint. Treat it as a\nserver key: keep it only in this function's secrets, never in an app, and rotate it like any\nserver key. The re-driver never reads the\nprice from the row (a signed-in user can edit their own row): the hold is the function's fixed\n`RENDER_CREDITS`, and a row whose `render_key` is not `owner:request_key` is failed, not started.\n(A hand-written row like that which also breaks the `render_key_shape` hook cannot be written at\nall, so the tick counts it as `unwritable` and it keeps one of the 20 slots: fix or delete it by\nhand.)\n\n## A contract test for your runtime\n\nEvery key your runtime posts in the `completed` body is written onto the row, so every key must be a\ndeclared field of `renders`. Keep that true in your own CI with a test beside your runtime's code. It\nfails the day someone adds a key to the completion body and forgets the field:\n\n```ts\n// render-contract.test.ts \u2014 vitest, in YOUR repo (the one holding vxil.config.ts)\nimport { describe, expect, it } from 'vitest';\nimport config from './vxil.config';\n// the bodies your runtime (or coordinator) really posts: import the builders from that code,\n// so the test follows it instead of a hand-copied list\nimport { completedBody, progressBody } from './runtime/callback-bodies';\n\ndescribe('render callbacks', () => {\n const declared = new Set(Object.keys(config.cms!.collections!.renders!.fields!));\n\n it('every key of the completion body is a declared renders field', () => {\n const body = completedBody({ objectId: 'obj_test', durationS: 1 });\n expect(Object.keys(body).filter((k) => !declared.has(k))).toEqual([]);\n });\n\n it('a progress ping carries only the keys the mirror keeps', () => {\n const ping = progressBody({ progress: 40, stage: 'encoding' });\n expect(Object.keys(ping).filter((k) => !['status', 'progress', 'stage', 'message'].includes(k))).toEqual([]);\n });\n});\n```\n\nThe platform also has a fallback for the day that test is missing. When the row refuses a completion\nmirror because of an undeclared key, vxil writes the status alone, so the row still says\n`completed`, and the run reports what it dropped (`GET /v1/jobs/runs/{run_id}` \u2192 `mirror_error`,\nwith `fields_dropped`). The app no longer hangs on `processing`, but the dropped keys are not on the\nrow. The test is what keeps them there.\n\n## The bounds to design against\n\n- **Deadline**: the run's timeout is clamped to `generation.maxTimeoutMs` \u2014 one hour at most. A render\n that can take longer should be split (the next step starts from `notify-ready`), or tracked on your own\n row without a platform-held reserve.\n- **In flight**: `generation.maxConcurrent` renders at once (20 here and by default; up to 200). Over\n it, `request-render` answers `429` and the row waits for `redrive-pending`, or for a retry with the\n same `request_key` ([the backlog](#the-backlog-maxconcurrent-429-and-the-re-driver)).\n- **Holds**: one render's reserve is clamped to `generation.maxReserveCredits` (50 here), and the sum of\n all open holds is capped by `generation.maxOutstandingReserveCredits` (5,000 here).\n- **Callback**: at most 256 KiB per post, JSON, every key a declared field of `renders`. About 120\n posts a minute per render plus a small allowance kept for the final post (both best-effort, per\n serving machine); progress at most every 5 s, and the final post retried until it lands\n ([callback limits](#callback-limits-progress-is-best-effort-the-final-post-must-arrive)).\n\n## Your token, and where it lives\n\n`render_url` and `render_token` are function secrets; `request-render` reads them at invoke time and\nputs them on the generation run, which vxil stores with the run (your project only) for as long as the\njobs retention keeps it. The token is **never returned by a read**: `GET /v1/jobs/runs/{run_id}` (and\nthe dashboard and MCP reads built on it) shows the provider header names with every value as\n`[redacted]`. Rotate it in both places (`vxil secrets set functions/render_token` and your endpoint's\nenv); runs already started keep the token they were started with.\n\n## Evidence\n\n- **Read in vxil's code**: the start request body (`provider.body` + `payload` + `callback_url`), the\n 20-second start bound, the retry ladder, the status mirror writing every completion key onto the row,\n the hold committed on `completed` and released on `failed` / the deadline, and `error` / `hint`\n becoming `error_class` / `error_hint` on `job.generation.failed`.\n- **Read in Trigger.dev's and Modal's documentation, not executed from vxil**: the trigger endpoint\n `POST https://api.trigger.dev/api/v1/tasks/{taskId}/trigger` with `{ payload, options }` and the\n `idempotencyKey` / `ttl` / `tags` / `machine` options; `task({ id, machine, maxDuration, retry, run,\n onFailure })` and `retry` options, `maxDuration` being CPU time per attempt with no `onFailure` when\n it is exceeded, `metadata.set`, and the `ffmpeg()` / `puppeteer()` build extensions; Trigger.dev\n Cloud's US-hosted operational data; Modal's `@modal.fastapi_endpoint` and `.spawn()`. Check each\n vendor's current docs before you ship.\n",
|
|
18605
|
+
"readme": "# Render Farm \u2014 long renders on a runtime you rent, with vxil holding the credits, the deadline and the callback\n\n```bash\nmkdir my-renders && cd my-renders\nvxil init --template render-farm # init scaffolds into the CURRENT directory\nprintf '%s' \"$RENDER_URL\" | vxil secrets set functions/render_url # your render endpoint (https)\nprintf '%s' \"$RENDER_TOKEN\" | vxil secrets set functions/render_token # a long random token it checks\n# the backlog re-driver's key: an API key of this backend holding ONLY jobs:write\nvxil login # keys mint needs a dashboard session, not a project key\nvxil keys mint --name render-redrive --scopes jobs:write --json | jq -r .api_key | vxil secrets set functions/vxil_jobs_key\nvxil push\n```\n\n> **Plan note.** The functions deploy on the Free plan when the project's workload is `staging` or\n> `development` (`vxil projects workload <slug> development`, or create it with\n> `vxil projects create <slug> --workload development`). On a Free `production` project, `vxil push` stops before it writes anything, naming the plan and the ways out: change the workload or upgrade to Developer, or run `vxil push --skip-functions` to apply the collections and config without the functions.\n> On the Free plan a function cron may also fire at most every 15 minutes, so change `redrive-pending`'s\n> schedule to `'*/15 * * * *'` there. The blueprint is written for Developer and up, where it runs every\n> minute. Everything else works the same; a backlog just drains more slowly.\n\nA video render, a transcode, a headless-browser capture: minutes of CPU, ffmpeg or Chromium. That does\nnot fit in a vxil function (a delivered trigger gets about a minute), and vxil will not grow a container\ntier or a workflow engine to run it. So the work runs on **a runtime you rent** \u2014 Trigger.dev, Modal,\na container on your own cloud account \u2014 and vxil keeps the four things that must survive while it\nruns:\n\n| vxil holds | so that |\n|---|---|\n| **the credits** reserved for the render | a failed, abandoned or cancelled render gives them back, and a user can never start more than they can pay for |\n| **the deadline** | a render your runtime never reports on fails and refunds after 30 minutes (at most one hour) |\n| **the signed completion callback** | your runtime needs no vxil key: the URL it is handed is the credential for that one render |\n| **the status row** | the app reads (or subscribes to) one `renders` row: `pending \u2192 processing \u2192 completed | failed`, plus everything your runtime sent back |\n\n**What this blueprint teaches that the others do not:** the hand-off to **your own** long-running\nruntime through a **webhook-mode generation run** \u2014 the contract your endpoint and your worker must\nkeep, and two complete runtime options below. (`fal-media` shows the same lane against a vendor queue\nAPI; `job-runner` shows a provider call vxil polls.)\n\n## What you get\n\n- **`renders`** \u2014 one row per render, owned by the user who asked for it (`strictEndUserScope`: a\n signed-in user reads only their own). `render_key` is the owner + `:` + the client's `request_key`\n (the composition is enforced by a `beforeWrite` hook) and is **unique**, so a double tap or a retried\n request finds the first row instead of starting a second render. Your runtime's answer lands on the\n row: `output_url`, `duration_s`, `progress`, `stage`, `message`.\n- **`request-render`** (http function, end-user mode) \u2014 creates-or-finds the row, then starts ONE\n generation run: your endpoint (`render_url`), your token on its `Authorization` header, a status\n mirror onto the row, `reserve_credits` for the render (5 `render_credits`), a 30-minute deadline, and\n the `render_key` as the run's `idempotency_key` \u2014 so a re-driven start gets the same run back. The\n credits it holds are the function's fixed price, never a value read from the row.\n- **`redrive-pending`** (cron function, every minute, `overlap: 'skip'`) \u2014 drains the backlog. When\n the project already has `generation.maxConcurrent` renders in flight, `request-render` answers `429`\n (with `queued: true`) and the row waits `pending` with no run. This function starts those rows,\n oldest first, with the **same** idempotency key, and tells the owner when one never starts\n ([the backlog](#the-backlog-maxconcurrent-429-and-the-re-driver)).\n- **`notify-ready`** (webhook function on `job.generation.`) \u2014 writes the failure cause onto the row (a\n platform class such as `GenerationExpired` gets a short human hint after it) and sends the owner one\n message per run (`Idempotency-Key: render-ready:<run_id>`). A render refused for too few credits\n keeps `insufficient_credits`; when `request-render` refused it, the caller already got the `402`\n and nothing is sent, and when the re-driver started it (the user last heard \"queued\"), the owner\n is told once (`Idempotency-Key: render-not-started:<item_id>`, the re-driver's own key). It\n **branches on `error_class`**, because not every `job.generation.failed` is a render that failed:\n `PaymentsUnavailable` (the credits could not be held at the start \u2014 the row is still queued and a\n new run follows) is skipped (and when a re-sent `request-render` raced that start and linked its\n run, the row is unlinked \u2014 `run_id: null`, only while it is still `pending` on that run \u2014 so\n `redrive-pending` starts it), and `Cancelled` (your own cancel) writes `error: cancelled` and sends\n nothing. Both are also lower-level events (`warn` / `info`), so they stay out of an immediate\n failure digest.\n- **credits** \u2014 a payments integration on the `mock` provider (no provider account needed to try it).\n \"Credits\" are usage units you meter, not money.\n\nThe row is a **view** for the app; the run and the ledger are the truth. If your app's client key\ncarries `cms:write`, a signed-in user can edit their own `renders` row (say, set `status` to\n`completed`) \u2014 that changes nothing they are charged or given. Anything that grants something on\ncompletion should read the run (`GET /v1/jobs/runs/{run_id}`) or react to `job.generation.completed`,\nas `notify-ready` does \u2014 or give the client key only `cms:read`.\n\n## The contract your runtime keeps\n\nWhatever runs the render, these are the only things it has to do:\n\n| Step | What arrives / what to send |\n|---|---|\n| **Start** | vxil `POST`s your `render_url` with `Authorization: Bearer <render_token>`, the header `x-vxil-run-id` and JSON `{ render_id, user_id, composition, props, payload: { generation_id, correlation_id, deadline_at }, callback_url }` (`user_id` is the render's owner). Check the token, **store or queue the work and answer `2xx` within 20 seconds** \u2014 never render inline, and never wait on a container's cold start. When you cannot take the work right now, say so at once with a `503` instead of holding the request open; the start is then sent again later, as after a timeout. `408` / `429` / `5xx` / a timeout is retried with backoff (30 s, then 1, 2 and 4 minutes, doubling up to 10 minutes; a `Retry-After` on your answer sets the wait, 1 s to 10 min, never past the deadline; `max_attempts` counts start calls) until the run's attempts are used \u2014 with the default five, the last try comes 7 to 8 minutes after the first, so leave room for that in the deadline. Any other `4xx` ends the run and refunds the credits. A start can arrive more than once (a lost answer is retried), so **dedupe by the run id** (`x-vxil-run-id`; in this blueprint `render_id` = `payload.generation_id` names the same single run), never by a hash of the content ([why](#one-render-one-run-dedupe-a-doubled-start)). |\n| **Progress** (optional, best-effort) | `POST callback_url` with `{\"status\": \"processing\", \"progress\": 40, \"stage\": \"encoding\"}`. A processing ping moves the row to `processing` and writes its `progress` (0\u2013100) / `stage` (\u2264 64 chars) / `message` (\u2264 200 chars) onto the row \u2014 `request-render` asks for that with `status_mirror.progress_fields` \u2014 so the app shows live progress by watching the row. Other keys on a ping are not written to the row (the run keeps the latest report, `GET /v1/jobs/runs/{run_id}` \u2192 `progress`). **At most one ping every 5 seconds**, sending the latest state; a ping that fails or answers `429` is simply dropped ([callback limits](#callback-limits-progress-is-best-effort-the-final-post-must-arrive)). |\n| **Done** (must arrive) | `POST callback_url` with `{\"status\": \"completed\", \"output_url\": \"https://\u2026\", \"duration_s\": 31.2, \"progress\": 100, \"stage\": \"done\"}` \u2014 or, when the output is uploaded into this project's files, `{\"status\": \"completed\", \"output_file\": \"obj_\u2026\", \"duration_s\": 31.2, \"progress\": 100, \"stage\": \"done\"}` ([below](#keys-stay-out-of-the-container-the-recommended-shape)). At most 256 KiB, and **every key a declared field of `renders`** (add a field before you send a new key; [test it](#a-contract-test-for-your-runtime)). The credits are committed and every key is written onto the row. **Retry this post on `429`, `5xx` and network errors, honouring `Retry-After`, until `deadline_at`** \u2014 never drop it (`postCallback` below). |\n| **Failed** (must arrive) | `POST callback_url` with `{\"status\": \"failed\", \"error\": \"render_failed\", \"hint\": \"ffmpeg exited 1: \u2026\"}`. The credits are refunded; `error` (a short code) and `hint` reach `notify-ready` as `error_class` / `error_hint` and land on the row's `error`. Retried exactly like `completed`. |\n| **Never answers** | the run fails at the deadline (`GenerationExpired`), the credits are refunded, and a callback after that changes nothing. |\n| **The deadline** | `payload.deadline_at` (an ISO time) is when vxil stops waiting \u2014 to the second: any callback that arrives at or after it ends the run as `GenerationExpired` (credits refunded), and a `completed` posted then is answered `{ generation_status: \"failed\", expired: true }` \u2014 the output exists, but the user was refunded and the row says failed. Check it before each attempt starts: a render that cannot finish by then should post `failed` and stop. |\n\n`callback_url` needs no other credential \u2014 and nothing else should see it. A repeated `completed` or\n`failed` post is answered with the settled state and changes nothing, so your worker can safely retry\nits own callback on a network error. The output bytes stay where your runtime wrote them (your bucket,\nyour CDN) and vxil stores the keys, not the file, unless a coordinator uploads the output into this\nproject's files ([next section](#keys-stay-out-of-the-container-the-recommended-shape)).\n\nEach start request also carries an `X-Vxil-Jobs-Signature` header (verifiable with your project's\njobs signing secret, `GET /v1/jobs/signing-secret`) and `x-vxil-run-id`. This blueprint uses the\nbearer token because it is one string comparison in any language.\n\n### Callback limits: progress is best-effort, the final post must arrive\n\nEach render's `callback_url` has its **own** budget at vxil's edge: about 120 posts a minute for that\none run. Every post over it, pings included, is counted against a small separate allowance (about 20\na minute), and on that allowance only a `completed` or `failed` post is accepted. A runtime that pings\nevery 5 seconds never gets near either number. One that sends more than about 140 posts in a minute\nuses the allowance up as well, and its final post waits for the next minute. All of a project's\ncallbacks together also share a ceiling of about 3,000 a minute, plus about 300 a minute for final posts.\nEvery number here is best-effort (counted per serving machine). The signature in the URL identifies\nthe run, so fifty renders posting from one shared egress IP do not share a budget. Over a budget the\nanswer is `429 rate_limited` with `Retry-After: 30`.\n\nTwo rules follow, and the helpers below keep both:\n\n- **Progress is best-effort.** Post at most one `processing` ping every 5 seconds, carrying the latest\n state; a ping that fails or answers `429` is dropped, and the next one carries the newer state.\n- **The final post must arrive.** A `completed` or `failed` post that answers `429`, `5xx` or never\n gets an answer is sent again, after `Retry-After` when there is one, until `deadline_at`. A repeat\n of a final post is harmless: it is answered with the settled state and changes nothing. Any other\n `4xx` means the URL or the body is wrong, so retrying will not help.\n\n```ts\n// callbacks.ts \u2014 for the coordinator, a task, or any worker that posts to callback_url\n/** completed / failed: MUST arrive. Retries 429, 5xx and network errors (honouring Retry-After)\n * until `untilMs`; throws only when it cannot deliver in time, or on another 4xx. */\nexport async function postCallback(callbackUrl: string, body: Record<string, unknown>, untilMs: number): Promise<void> {\n for (let attempt = 0; ; attempt++) {\n let res: Response | undefined;\n try {\n res = await fetch(callbackUrl, {\n method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body),\n });\n } catch { /* a network error: send it again */ }\n if (res && res.ok) return;\n if (res && res.status !== 429 && res.status < 500) throw new Error(`callback refused: ${res.status}`);\n const retryAfterS = Number(res?.headers.get('retry-after'));\n const waitMs = retryAfterS > 0 ? retryAfterS * 1000 : Math.min(1000 * 2 ** attempt, 30_000);\n if (Date.now() + waitMs > untilMs) throw new Error(`callback not delivered in time (last answer: ${res?.status ?? 'none'})`);\n await new Promise((r) => setTimeout(r, waitMs));\n }\n}\n\n/** processing: best-effort, at most one post every `gapMs`; a refused or failed ping is dropped. */\nexport function progressReporter(callbackUrl: string, gapMs = 5_000) {\n let last = 0;\n return async (p: { progress?: number; stage?: string; message?: string }): Promise<void> => {\n if (Date.now() - last < gapMs) return; // coalesced: the next ping carries newer state\n last = Date.now();\n await fetch(callbackUrl, {\n method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify({ status: 'processing', ...p }),\n }).catch(() => undefined);\n };\n}\n```\n\n### One render, one run: dedupe a doubled start\n\nvxil sends the start again when it did not get your answer: a timeout, a dropped connection, a `5xx`.\nSo the same start can reach you twice, even while the first copy is still being handled. Claim the\nwork under the **run id** (`x-vxil-run-id`) before you launch anything. Answer `202` to a repeat\nonly once the work is launched. While the first copy is still launching, answer `503`, so the start\nstays open and vxil sends it again; a `202` would end vxil's retries even if that first launch then failed. In this blueprint `render_id` (= `payload.generation_id`, the row's id)\nnames the same single run, because `request-render` starts one run per row.\n\nNever key the claim by a hash of the content (the composition and props). Two users who ask for the\nsame render produce two runs with one hash: the second start would find the first's claim and be\nrefused, or overwrite the first's `callback_url`, and that run would wait out its deadline and refund.\nIf you want identical renders to share their output, claim by run id and look the output up by hash\nas a separate step.\n\n## Keys stay out of the container (the recommended shape)\n\nThe render container is the part of your system that runs the most third-party code (ffmpeg,\nChromium, fonts and media from the user's props), so give it **no vxil key at all**. This is the\nshape the blueprint recommends, whatever runs the container:\n\n```\nvxil \u2500\u2500start\u2500\u2500\u25B6 coordinator \u2500\u2500launch\u2500\u2500\u25B6 container\n (render_url) \u2502 progress / failed \u2500\u2500\u2500\u2500\u2500\u2500\u25B6 callback_url (keyless)\n \u25B2 output bytes \u2500\u2500\u2500\u2500\u2500\u2518\n \u2502\n \u2514\u2500 mints the upload URL with ITS files:write key, PUTs the bytes,\n completes the object, POSTs \"completed\" + output_file \u2500\u2500\u25B6 callback_url\n```\n\n- **The files feature is on.** The blueprint's config enables it (`files: { enabled: true }`) and\n declares `output_file` as a `file` field; without it every upload-url call is refused and the\n render waits out its deadline. Its per-object ceiling, `maxObjectBytes`, is 100 MB by default,\n enough for the buffered coordinator below; raise it for long or high-bitrate renders.\n- **The coordinator** is your `render_url`: a small endpoint in your own account \u2014 an edge worker\n with a per-render lock, or a route on any thin server. It is the only piece that holds a vxil key,\n and that key holds only **`files:write`** (`vxil keys mint --name render-uploads --scopes files:write`).\n- **The container** gets the job, the keyless `callback_url` (for `processing` pings and for\n `failed`), an `output_url` on the coordinator and an **output ticket** for it: a token signed for\n that one render and useless after its deadline, sent on the `Authorization` header (never in the\n URL, where access logs would keep it).\n- **The upload happens once the size is known.** The container POSTs the finished file to its\n ticket. The coordinator reads it, mints the files upload URL for exactly that size\n (the quota pre-check uses it), PUTs the bytes, completes the object, and only then posts\n `completed` with the object id as `output_file`. A crash anywhere before that leaves the render\n open, and it refunds at the deadline like any other.\n\n```ts\n// coordinator.ts \u2014 your render_url. A standard fetch handler (an edge worker, or a Node 18+ adapter).\n// Env: RENDER_TOKEN (= the vxil secret render_token), TICKET_SECRET (a long random string),\n// VXIL_FILES_KEY (an API key holding ONLY files:write), VXIL_BASE (https://api.vxil.com).\nimport { postCallback } from './callbacks';\n\ntype Start = {\n render_id: string; user_id: string; composition: string; props?: Record<string, unknown>;\n payload: { generation_id: string; deadline_at: string }; callback_url: string;\n};\ntype Ticket = { run_id: string; render_id: string; user_id: string; callback_url: string; exp: number };\ntype Env = { RENDER_TOKEN: string; TICKET_SECRET: string; VXIL_FILES_KEY: string; VXIL_BASE: string };\n\nconst enc = new TextEncoder();\nconst b64u = (b: ArrayBuffer | Uint8Array) =>\n btoa(String.fromCharCode(...new Uint8Array(b))).replace(/\\+/g, '-').replace(/\\//g, '_').replace(/=+$/, '');\nconst unb64u = (s: string) => Uint8Array.from(atob(s.replace(/-/g, '+').replace(/_/g, '/')), (c) => c.charCodeAt(0));\nconst hmacKey = (secret: string) =>\n crypto.subtle.importKey('raw', enc.encode(secret), { name: 'HMAC', hash: 'SHA-256' }, false, ['sign', 'verify']);\nasync function sealTicket(env: Env, t: Ticket): Promise<string> {\n const body = b64u(enc.encode(JSON.stringify(t)));\n return `${body}.${b64u(await crypto.subtle.sign('HMAC', await hmacKey(env.TICKET_SECRET), enc.encode(body)))}`;\n}\nasync function openTicket(env: Env, raw: string): Promise<Ticket | null> {\n const [body, sig] = raw.split('.');\n if (!body || !sig) return null;\n try {\n // crypto.subtle.verify compares in constant time (a `!==` on the signature would not)\n if (!(await crypto.subtle.verify('HMAC', await hmacKey(env.TICKET_SECRET), unb64u(sig), enc.encode(body)))) return null;\n const t = JSON.parse(new TextDecoder().decode(unb64u(body))) as Ticket;\n return t.exp > Date.now() ? t : null;\n } catch {\n return null; // not base64url / not JSON\n }\n}\n/** The output's file extension, from the Content-Type the container sends. */\nconst EXT: Record<string, string> = {\n 'video/mp4': 'mp4', 'video/webm': 'webm', 'image/gif': 'gif', 'image/png': 'png', 'image/jpeg': 'jpg', 'application/pdf': 'pdf',\n};\n/** Sent with a 503: when to send the request again. */\nconst AGAIN_IN_30S = { 'retry-after': '30' };\nconst vxil = (env: Env, path: string, body?: unknown) => fetch(`${env.VXIL_BASE}${path}`, {\n method: 'POST',\n headers: { authorization: `Bearer ${env.VXIL_FILES_KEY}`, 'content-type': 'application/json' },\n ...(body ? { body: JSON.stringify(body) } : {}),\n});\n\nexport default {\n async fetch(req: Request, env: Env): Promise<Response> {\n const url = new URL(req.url);\n\n // 1. the start: check the token, launch the container, answer inside 20 s\n if (req.method === 'POST' && url.pathname === '/start') {\n if (req.headers.get('authorization') !== `Bearer ${env.RENDER_TOKEN}`) return new Response('unauthorized', { status: 401 });\n const s = (await req.json()) as Start;\n // a run started before request-render sent user_id (an older copy of this\n // blueprint): a server-mode upload must name its user, so refuse the start \u2014\n // a 4xx ends that run and refunds it, instead of a 422 at upload time\n if (!s.user_id) return new Response('start body has no user_id: redeploy request-render', { status: 400 });\n // one run, one render: claim the work under the RUN id before launching anything. The claim\n // says `launching` (and expires after a minute) until the launch succeeds, then `launched`.\n // A start vxil re-sends (a lost answer) is acknowledged with 202 only once the work is\n // `launched`; while another copy is still launching it gets 503, so vxil keeps retrying\n // instead of treating a launch that may yet fail as accepted.\n // Never claim by a content hash: two users' identical renders are two runs.\n const runId = req.headers.get('x-vxil-run-id') ?? s.payload.generation_id;\n const claimKey = `start:${runId}`;\n if (!(await store.claim(claimKey, 'launching', 60_000))) {\n if ((await store.get(claimKey)) === 'launched') return Response.json({ accepted: true, duplicate: true }, { status: 202 });\n return new Response('this render is still being launched', { status: 503, headers: AGAIN_IN_30S });\n }\n const ticket = await sealTicket(env, {\n run_id: runId, render_id: s.render_id, user_id: s.user_id, callback_url: s.callback_url,\n exp: Date.parse(s.payload.deadline_at), // useless once vxil stops waiting\n });\n try {\n await launchContainer({ // YOUR container platform's API: it must QUEUE\n run_id: runId, render_id: s.render_id, // the job and return at once \u2014 never wait.\n // Pass run_id as its idempotency key if it has one\n composition: s.composition, props: s.props ?? {},// here for a cold start\n deadline_at: s.payload.deadline_at,\n callback_url: s.callback_url, // for processing pings and `failed`\n output_url: `${url.origin}/output`, // POST the file here, with\n output_ticket: ticket, // Authorization: Bearer <output_ticket>\n });\n } catch {\n await store.release(claimKey); // not launched: let vxil's retry try again\n return new Response('cannot take the render right now', { status: 503, headers: AGAIN_IN_30S });\n }\n await store.put(claimKey, 'launched'); // no expiry: every later copy is a duplicate\n return Response.json({ accepted: true }, { status: 202 });\n }\n\n // 2. the output: the container POSTs the finished file here (again, on any non-2xx answer)\n if (req.method === 'POST' && url.pathname === '/output') {\n const t = await openTicket(env, (req.headers.get('authorization') ?? '').replace(/^Bearer /, ''));\n if (!t) return new Response('bad or expired ticket', { status: 403 });\n const contentType = (req.headers.get('content-type') ?? '').split(';')[0]!.trim().toLowerCase();\n const ext = EXT[contentType];\n if (!ext) return new Response(`send the output's Content-Type (one of: ${Object.keys(EXT).join(', ')})`, { status: 415 });\n // a re-sent output after the upload already happened: skip straight to the settle\n let object_id = await store.get(`output:${t.run_id}`);\n if (!object_id) {\n // buffered: fine for outputs of tens of MB \u2014 see \"Very large outputs\" below\n const bytes = await req.arrayBuffer();\n const size = bytes.byteLength;\n if (size === 0) return new Response('empty output', { status: 400 });\n\n const minted = await vxil(env, '/v1/files/upload-url', {\n user_id: t.user_id, filename: `${t.render_id}.${ext}`, content_type: contentType, size_bytes: size,\n });\n if (!minted.ok) return new Response(`upload-url ${minted.status}`, { status: 502 }); // the container sends the file again\n const m = ((await minted.json()) as { data: { object_id: string; upload_url: string } }).data;\n const put = await fetch(m.upload_url, { method: 'PUT', headers: { 'content-type': contentType }, body: bytes });\n if (!put.ok) return new Response(`upload ${put.status}`, { status: 502 });\n const done = await vxil(env, `/v1/files/${encodeURIComponent(m.object_id)}/complete`);\n if (!done.ok) return new Response(`complete ${done.status}`, { status: 502 });\n object_id = m.object_id;\n await store.put(`output:${t.run_id}`, object_id);\n }\n\n // 3. settle the render on the keyless callback: a MUST-ARRIVE post. Retry here for up to a\n // minute (never past the deadline); if it still did not land, answer 503 and the container\n // sends the output again \u2014 the note above turns that into one more settle attempt.\n try {\n await postCallback(t.callback_url,\n { status: 'completed', output_file: object_id, progress: 100, stage: 'done' },\n Math.min(t.exp, Date.now() + 60_000));\n } catch (e) {\n return new Response(`callback: ${String(e)}`, { status: 503, headers: AGAIN_IN_30S });\n }\n return Response.json({ object_id });\n }\n return new Response('not found', { status: 404 });\n },\n};\n\ndeclare function launchContainer(job: Record<string, unknown>): Promise<void>; // your container platform's API\n/** Your coordinator's own small store: a key-value namespace, a table, a per-key lock. `claim` is\n * an insert-if-absent (of `value`, expiring after `ttlMs`) that answers true for the FIRST caller\n * only; `put` writes without an expiry. */\ndeclare const store: {\n claim(key: string, value: string, ttlMs: number): Promise<boolean>; release(key: string): Promise<void>;\n get(key: string): Promise<string | null>; put(key: string, value: string): Promise<void>;\n};\n```\n\nThe container's side is three kinds of HTTP call and no vxil key: `processing` pings to\n`callback_url`, the file to `output_url` (with `Authorization: Bearer <output_ticket>` and the\nfile's `Content-Type`), and `{\"status\": \"failed\", \u2026}` to `callback_url` if it gives up. The\ncontainer sends its pings through `progressReporter` and its `failed` through `postCallback`\n([callbacks.ts](#callback-limits-progress-is-best-effort-the-final-post-must-arrive)), and it sends\nthe output again on any non-`2xx` answer until `deadline_at`. Five notes on the shape:\n\n- **A doubled start launches one container.** The coordinator claims `start:<run id>` as\n `launching` before it launches and marks it `launched` once the launch succeeded. A start vxil\n re-sends is answered `202` and launches nothing only when the claim says `launched`; while the\n first copy is still launching, the re-sent one is answered `503`, so vxil keeps the start alive.\n (A `202` there would end vxil's retries, and if the first launch then failed nothing would run\n and the render would refund at its deadline.) When the launch fails, the coordinator releases\n the claim and answers `503` at once, and vxil sends the start again later. A coordinator that\n dies mid-launch leaves a `launching` claim that expires after a minute, so a later copy launches;\n pass the run id as your container platform's idempotency key, where it has one, so that copy\n cannot start a second container if the first launch did go through.\n- **A retried output post** (the container saw a network error after the coordinator had uploaded)\n finds `output:<run id>` in the coordinator's store, skips the upload and only sends the\n `completed` post again. A repeated `completed` is answered with the settled state and changes\n nothing, so the final post can be retried as often as it takes.\n- **Very large outputs.** The coordinator above holds the whole file in memory, and passing gigabytes\n through it costs its bandwidth too. For big files, have the container report the size first\n (`POST /output-url` with its ticket on the `Authorization` header and `{ size_bytes }`) and let the coordinator answer with the\n presigned upload URL it minted; the container PUTs straight to it, and the coordinator completes\n the object and posts `completed` when the container says it is done. The container still holds no\n key: a presigned URL is good for one object for a few minutes.\n- **Upgrading an earlier copy of this blueprint.** Renders started before `request-render` put\n `user_id` in the start body have none, and a server-mode upload must name its user. The\n coordinator refuses such a start with a `400`, which ends that run and refunds it at once;\n redeploy `request-render` (`vxil push`) before you point `render_url` at the coordinator.\n- **Serving it.** The row now holds a files object id. Read it back with a signed download URL, or\n publish it from a settle function with a server key (`vx.files.publish(object_id)`) for a stable\n public URL served from the edge cache.\n\nOptions A and B below show the same contract with the runtime posting `completed` itself (an\n`output_url` in your own bucket). Either can adopt the coordinator: point `render_url` at it, and have\nstep 1 trigger the Trigger.dev task or spawn the Modal function.\n\n## Option A \u2014 Trigger.dev (v4)\n\nTrigger.dev runs the render as a task on a machine you pick, with ffmpeg or Chromium baked into the\nimage, retries, and its own run dashboard. Two pieces: a **relay endpoint** that turns vxil's start\nrequest into a Trigger.dev trigger, and the **task**.\n\n**Why a relay, and not `render_url` pointed straight at Trigger.dev's trigger API?** vxil sends\n`callback_url` beside `payload` at the top level of the body, and Trigger.dev's trigger API passes\nonly `payload` to the task \u2014 the task would never see where to report. The relay is ~30 lines and is\nalso where your `render_token` is checked. Host it anywhere that serves https: a serverless function on\nyour web host, a small edge worker, a route in your existing API.\n\n```ts\n// relay.ts \u2014 your render_url. Standard fetch handler (edge worker / serverless function / Node 18+ adapter).\n// Env: RENDER_TOKEN (the same value as the vxil secret render_token), TRIGGER_SECRET_KEY (tr_prod_\u2026 / tr_dev_\u2026).\ntype Start = {\n render_id: string; composition: string; props?: Record<string, unknown>;\n payload: { generation_id: string; correlation_id?: string; deadline_at: string }; callback_url: string;\n};\n\nexport default {\n async fetch(req: Request, env: { RENDER_TOKEN: string; TRIGGER_SECRET_KEY: string }): Promise<Response> {\n if (req.method !== 'POST') return new Response('method not allowed', { status: 405 });\n if (req.headers.get('authorization') !== `Bearer ${env.RENDER_TOKEN}`) {\n return new Response('unauthorized', { status: 401 }); // a 4xx ends the vxil run (and refunds)\n }\n const s = (await req.json()) as Start;\n const res = await fetch('https://api.trigger.dev/api/v1/tasks/render-video/trigger', {\n method: 'POST',\n headers: { authorization: `Bearer ${env.TRIGGER_SECRET_KEY}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n payload: {\n render_id: s.render_id, composition: s.composition, props: s.props ?? {},\n callback_url: s.callback_url, deadline_at: s.payload.deadline_at,\n },\n options: {\n // a start vxil re-sends (a lost answer) triggers the SAME Trigger.dev run\n idempotencyKey: `render:${s.render_id}`,\n // a render still queued after 3 minutes is dropped (it never runs, and no\n // onFailure fires \u2014 vxil refunds it at its deadline). Part of the budget below.\n ttl: '3m',\n tags: [`render_${s.render_id}`],\n },\n }),\n });\n if (res.ok) return Response.json({ accepted: true }, { status: 202 });\n // Trigger.dev busy or down: answer 503 and vxil tries the start again; anything else ends the run\n return new Response(`trigger.dev ${res.status}`, { status: res.status === 429 || res.status >= 500 ? 503 : 400 });\n },\n};\n```\n\n```ts\n// trigger.config.ts \u2014 ffmpeg and Chromium in the task image\nimport { defineConfig } from '@trigger.dev/sdk';\nimport { ffmpeg } from '@trigger.dev/build/extensions/core';\nimport { puppeteer } from '@trigger.dev/build/extensions/puppeteer';\n\nexport default defineConfig({\n project: '<your project ref>',\n dirs: ['./trigger'],\n maxDuration: 600, // CPU seconds PER ATTEMPT \u2014 the task sets its own; see the budget below\n build: { extensions: [ffmpeg(), puppeteer()] }, // puppeteer also needs PUPPETEER_EXECUTABLE_PATH set in the Trigger.dev env\n});\n```\n\n```ts\n// trigger/render-video.ts \u2014 the task: render, upload, report to vxil\nimport { task, metadata, logger } from '@trigger.dev/sdk';\nimport { postCallback, progressReporter } from '../callbacks'; // the helpers above\n\ntype Payload = {\n render_id: string; composition: string; props: Record<string, unknown>;\n callback_url: string; deadline_at: string; // when vxil stops waiting (ISO)\n};\n\n/** The longest one attempt takes, wall clock, with margin. An attempt that cannot\n * finish before deadline_at does not start: it tells vxil, so the credits come back now. */\nconst ATTEMPT_WALL_MS = 12 * 60_000;\n\n/** completed / failed: retried on 429 / 5xx / network errors (Retry-After honoured) until the\n * deadline \u2014 a repeated final post is a no-op on vxil's side, so retrying is always safe. */\nconst report = (p: Payload, body: Record<string, unknown>) =>\n postCallback(p.callback_url, body, Date.parse(p.deadline_at));\n\nexport const renderVideo = task({\n id: 'render-video',\n machine: 'large-1x', // 4 vCPU / 8 GB \u2014 size to your renders\n maxDuration: 600, // CPU seconds per attempt (10 min) \u2014 not wall time\n retry: { maxAttempts: 2, minTimeoutInMs: 5_000, maxTimeoutInMs: 30_000 },\n run: async (p: Payload) => {\n if (Date.now() + ATTEMPT_WALL_MS > Date.parse(p.deadline_at)) {\n // too late to finish inside vxil's deadline: refund now, and do not retry\n await report(p, { status: 'failed', error: 'deadline', hint: 'no time left for another attempt' });\n return { skipped: 'deadline' };\n }\n metadata.set('stage', 'rendering'); // Trigger.dev's own run view\n const progress = progressReporter(p.callback_url); // best-effort, at most one ping per 5 s\n await progress({ stage: 'rendering', progress: 0 });\n\n // \u2026your render: drive Chromium for frames, run ffmpeg \u2014 call progress({ progress, stage })\n // as often as you like (it coalesces) \u2014 and write the file\n // to YOUR bucket keyed by render_id (so a retried attempt overwrites, not duplicates)\u2026\n const outputUrl = `https://cdn.example.com/renders/${p.render_id}.mp4`;\n const durationS = 31.2;\n logger.info('rendered', { render_id: p.render_id, outputUrl });\n if (Date.now() > Date.parse(p.deadline_at)) {\n // vxil has already failed and refunded this render: the post below is answered\n // with that settled state. Your sizing is off \u2014 widen the budget below.\n logger.warn('finished after the vxil deadline', { render_id: p.render_id });\n }\n\n await report(p, {\n status: 'completed', output_url: outputUrl, duration_s: durationS, progress: 100, stage: 'done',\n });\n return { output_url: outputUrl };\n },\n // after the last attempt THROWS: tell vxil, so the credits come back now, not at the\n // deadline. Not called when an attempt exceeds maxDuration or the run expires on its\n // ttl \u2014 those refund only at vxil's deadline.\n onFailure: async ({ payload, error }) => {\n await report(payload, {\n status: 'failed', error: 'render_failed', hint: String(error instanceof Error ? error.message : error).slice(0, 200),\n });\n },\n});\n```\n\n**Budget the wall clock.** Trigger.dev's limits and vxil's deadline are separate clocks, and only\nvxil's refunds. Size them so a render always ends \u2014 `completed` or `failed` \u2014 before vxil's deadline:\n\n```\nttl + maxAttempts \xD7 (longest attempt, wall clock) + retry backoff < timeout.after_ms\n3 min + 2 \xD7 12 min + \u2264 1 min = 28 min < 30 min\n```\n\n`maxDuration` counts **CPU time per attempt**, not wall time across the run, so it does not bound\nthe sum: the `deadline_at` check at the start of each attempt does. If your renders need more, raise\n`RENDER_DEADLINE_MS` in `request-render` (up to `generation.maxTimeoutMs`, one hour) and resize the\nrest to fit.\n\n**Be honest with yourself about three things before you ship on Trigger.dev Cloud:**\n\n- **Data residency.** Trigger.dev Cloud keeps its operational and log data \u2014 including each run's\n payload \u2014 in the US (us-east-1), even when the machines run elsewhere. Your render props and the\n `callback_url` pass through it. If that rules it out, self-host Trigger.dev for your own app, or use\n option B.\n- **The callback URL is a credential for one render.** It appears in Trigger.dev's run payload and\n dashboard. It can settle only that render, and stops mattering once the render is settled.\n- **Two clocks, and `onFailure` is not a guarantee.** Trigger.dev calls `onFailure` only after the last\n attempt throws. A run that exceeds `maxDuration`, or expires on its `ttl` before it starts, ends\n without it \u2014 vxil refunds those at its deadline, not sooner. Keep the budget above, and keep the\n `deadline_at` check, so a late attempt refunds early instead of finishing after the refund.\n\n## Option B \u2014 your own container runtime (Modal, Fly, a container on your own cloud account)\n\nSame contract, no relay: the endpoint you deploy **is** `render_url`. It must answer within 20 seconds,\nso it only checks the token, hands the job to a background worker and answers `2xx`; the worker renders\nand posts back. On Modal:\n\n```python\n# render_app.py \u2014 `modal deploy render_app.py`; render_url = the endpoint's https URL\nimport http.client, json, os, time, urllib.error, urllib.request\nfrom datetime import datetime, timedelta, timezone\nimport modal\nfrom fastapi import HTTPException, Request # also `pip install fastapi` where you run `modal deploy`\n\nimage = (modal.Image.debian_slim()\n .apt_install(\"ffmpeg\", \"chromium\")\n .pip_install(\"fastapi[standard]\"))\napp = modal.App(\"render-farm\", image=image)\nsecrets = [modal.Secret.from_name(\"render-farm\")] # RENDER_TOKEN\n\ndef _post(callback_url: str, body: dict) -> None:\n req = urllib.request.Request(callback_url, data=json.dumps(body).encode(),\n headers={\"content-type\": \"application/json\"}, method=\"POST\")\n with urllib.request.urlopen(req, timeout=30) as res:\n res.read()\n\ndef report(callback_url: str, body: dict, deadline: datetime) -> None:\n \"\"\"completed / failed: MUST arrive. Retries 429, 5xx and network errors (honouring\n Retry-After) until the deadline; a repeated final post changes nothing on vxil's side.\"\"\"\n attempt = 0\n while True:\n wait = min(2 ** attempt, 30)\n attempt += 1\n try:\n _post(callback_url, body)\n return\n except urllib.error.HTTPError as e:\n if e.code != 429 and e.code < 500:\n raise # another 4xx: the URL or the body is wrong\n ra = e.headers.get(\"retry-after\")\n wait = int(ra) if ra and ra.isdigit() else wait\n except (OSError, http.client.HTTPException):\n pass # no answer, or the connection dropped mid-answer\n # (URLError, a reset, RemoteDisconnected, IncompleteRead,\n # a timeout): send it again\n if datetime.now(timezone.utc) + timedelta(seconds=wait) > deadline:\n raise RuntimeError(\"callback not delivered before the deadline\")\n time.sleep(wait)\n\n_last_ping = {}\ndef ping(callback_url: str, body: dict) -> None:\n \"\"\"processing: best-effort, at most one every 5 s; a refused or failed ping is dropped.\"\"\"\n now = time.monotonic()\n if now - _last_ping.get(callback_url, 0.0) < 5:\n return\n _last_ping[callback_url] = now\n try:\n _post(callback_url, {\"status\": \"processing\", **body})\n except Exception:\n pass\n\nATTEMPT_WALL_S = 25 * 60 # = the timeout below; a call cut off there may never reach its except\n\n@app.function(cpu=4, memory=8192, timeout=ATTEMPT_WALL_S, secrets=secrets)\ndef render(job: dict) -> None:\n cb = job[\"callback_url\"]\n deadline = datetime.fromisoformat(job[\"payload\"][\"deadline_at\"].replace(\"Z\", \"+00:00\"))\n if datetime.now(timezone.utc) + timedelta(seconds=ATTEMPT_WALL_S) > deadline:\n # queued too long to finish before vxil stops waiting: refund now\n report(cb, {\"status\": \"failed\", \"error\": \"deadline\", \"hint\": \"started too late to finish\"}, deadline)\n return\n try:\n ping(cb, {\"stage\": \"rendering\", \"progress\": 0})\n # \u2026render with ffmpeg / chromium (ping(cb, {...}) as often as you like),\n # upload to YOUR bucket keyed by job[\"render_id\"]\u2026\n output_url = f\"https://cdn.example.com/renders/{job['render_id']}.mp4\"\n except Exception as e: # the RENDER failed: tell vxil now, so the credits come back before the deadline\n report(cb, {\"status\": \"failed\", \"error\": \"render_failed\", \"hint\": str(e)[:200]}, deadline)\n raise\n # the output exists: only `completed` may follow. Outside the try on purpose, so a\n # trouble delivering it can never turn into a `failed` post that refunds a finished render.\n report(cb, {\"status\": \"completed\", \"output_url\": output_url, \"duration_s\": 31.2,\n \"progress\": 100, \"stage\": \"done\"}, deadline)\n\n@app.function(secrets=secrets)\n@modal.fastapi_endpoint(method=\"POST\")\nasync def start(request: Request):\n if request.headers.get(\"authorization\") != f\"Bearer {os.environ['RENDER_TOKEN']}\":\n raise HTTPException(status_code=401, detail=\"unauthorized\") # a 4xx ends the vxil run\n job = await request.json()\n await render.spawn.aio(job) # queued; returns at once, well inside the 20-second window\n return {\"accepted\": True}\n```\n\n`spawn` queues the call and returns immediately, so the endpoint answers in well under a second. A\nstart can arrive twice (a lost answer is retried), so dedupe on the run id\n(`request.headers[\"x-vxil-run-id\"]`; never a hash of the props): keep the run ids you have spawned in\na `modal.Dict`, or make the render overwrite the same output key.\n\nAnything else that can (1) answer an https POST in under 20 s, (2) run the work in the background and\n(3) POST JSON to a URL fits the same contract: a Fly Machine started per job, a container service with\na queue in front, your own GPU box. vxil does not care what runs the render \u2014 only that the start is\nacknowledged quickly and the callback eventually comes.\n\n## Run it\n\nGive a user some credits from your server (or sell `render_pack_100` through your payments provider):\n\n```bash\ncurl -s -X POST \"https://api.vxil.com/v1/payments/credits/grant\" \\\n -H \"authorization: Bearer $KEY\" -H 'content-type: application/json' \\\n -H 'idempotency-key: welcome-u1' \\\n -d '{\"user_id\":\"<the user id>\",\"credit_type\":\"render_credits\",\"amount\":25,\"source\":\"welcome\"}'\n```\n\nStart a render **with the user's session** (end-user mode \u2014 the held credits are forced onto that\nuser):\n\n```ts\nimport { Vxil } from '@vxil/sdk';\n\n// after `vxil gen`, vx.fn['request-render'] is typed from the function's declared signature\nconst vx = new Vxil({ apiKey: process.env.VXIL_PUBLISHABLE_KEY!, endUserToken: process.env.USER_SESSION! });\nconst started = await vx.fn['request-render']({\n composition: 'promo-30s', request_key: 'promo-1', props: { headline: 'Spring sale' },\n});\n// \u2192 { item_id, run_id, credits: 5 }\n// (or { duplicate: true, request_key, item_id, run_id, status } on a retry)\n```\n\nThen read the row \u2014 or subscribe to its changes \u2014 until `status` is `completed`:\n\n```ts\nif ('item_id' in started) {\n const row = await vx.from('renders').get(started.item_id);\n // row.status \u2192 'completed', row.output_url \u2192 'https://cdn.example.com/renders/\u2026.mp4'\n}\n```\n\n**Try it before you have a runtime.** Point `render_url` at any https endpoint that answers `2xx`\n(a request-bin works) and play the runtime yourself: copy `callback_url` from the request it received,\nthen\n\n```bash\ncurl -s -X POST \"$CALLBACK_URL\" -H 'content-type: application/json' \\\n -d '{\"status\":\"completed\",\"output_url\":\"https://cdn.example.com/x.mp4\",\"duration_s\":12.5,\"progress\":100,\"stage\":\"done\"}'\n# \u2192 { \"data\": { \"run_id\": \"run_\u2026\", \"generation_status\": \"completed\" } } \u2014 and the row says so\n```\n\n## How it fails, and what the user sees\n\n| what happened | the run | the row | the credits |\n|---|---|---|---|\n| runtime posted `completed` | `completed` | `status: completed` + every key it sent | committed |\n| runtime posted `failed` | `failed` | `status: failed`, `error` = its code + hint (written by `notify-ready`) | refunded |\n| runtime never called back | `failed` (`GenerationExpired`) at the deadline | `failed`, `error: GenerationExpired: the render did not finish before its deadline` | refunded |\n| runtime finished after the deadline | `failed` (`GenerationExpired`) \u2014 exact to the second; the late `completed` is answered `failed` / `expired: true` | `failed` | refunded (your compute was spent \u2014 budget the clocks) |\n| the final post was answered `429` (or `5xx`, or got no answer) | still open: nothing is settled until a post lands | unchanged | still held \u2014 `postCallback` sends it again after `Retry-After`; give up only at `deadline_at`, when the run refunds anyway |\n| a progress ping was answered `429` | unchanged | the previous progress stays | unchanged \u2014 drop the ping; the next one carries the newer state |\n| your endpoint answered `5xx` / timed out | start retried with backoff; terminal after the attempts | `processing` \u2192 `failed`, `error: RetryableHttp: the render endpoint kept failing to accept the render (retries exhausted)` (or `NetworkError: \u2026` when it could not be reached) | held until then, then refunded |\n| your endpoint answered another `4xx` (a bad token) | `failed` at once | `failed` | refunded |\n| the user had too few credits (at `request-render`) | ended at once (`ReserveInsufficient`), never started | `failed`, `error: insufficient_credits`; that `request_key` is spent; the caller got the `402`, no message | nothing held |\n| too many renders in flight, payments briefly unreachable, or a jobs-side fault | not created yet (the caller gets `429` / `503` with `queued: true` and `retry_after`, or `502` with `queued: true`) | `pending`, no run \u2014 **queued**: `redrive-pending` starts it when a slot frees up (or the app calls again with the **same** `request_key`) | held when it starts |\n| the user had too few credits when the re-driver started it | ended at once (`ReserveInsufficient`) | `failed`, `error: insufficient_credits`; the owner is told once (they last heard \"queued\") | nothing held |\n| still no free slot an hour later | never created | `failed`, `error: not_started: the render waited too long for a free slot` (written by `redrive-pending`); the owner is told once | nothing held |\n| the re-driver's start was refused (`400` / `401` / `403` / `422`: a non-https or private `render_url`, a wrong or revoked `vxil_jobs_key`) | not created | still `pending` \u2014 the tick stops and reports it (`stopped: { status, code }` in the function's logs); fix the setup and the next tick carries on; the one-hour bound still applies | nothing held |\n\n## The backlog: `maxConcurrent`, `429` and the re-driver\n\n`generation.maxConcurrent` is how many generation runs this project may have **open** at once: 20 by\ndefault, settable up to 200 in the jobs config. This blueprint sets 20; raise it to what your plan\nand your runtime can carry:\n\n```ts\njobs: { enabled: true, generation: { maxConcurrent: 50 /* 1\u2013200, default 20 */ } },\n```\n\nAt the cap a new start is refused with `429` and `Retry-After: 5`. **No run is created and nothing\nis held yet.** The same holds when the credits cannot be held because payments is briefly\nunreachable: the start is refused with `503 payments_unavailable` (nothing started, nothing held;\nit names a 15-second wait), `request-render` answers `503` with `queued: true`, and the re-driver starts\nthe row on a later tick with the same key (a fresh run). Never a free render: that only happens if\nyou opt in with `reserve_credits.on_unavailable: 'proceed'`. `request-render` passes the `429` on to the app with `queued: true` (and the\n`item_id` and `request_key`), and leaves the row `pending` with no `run_id`. That render is\n**queued, not refused**: the re-driver will start it, and hold its credits then, up to an hour\nlater. So the app must treat a `429` with `queued: true` as \"queued\" \u2014 show it, watch the row \u2014 and\nmust **never retry it with a new `request_key`**: that is a second render, and both are charged. A\nretry with the **same** `request_key` is always safe. Rows like that are the backlog, and two\nthings drain it:\n\n1. **`redrive-pending`, every minute.** It reads the oldest `pending` rows with no run that are at\n least 30 s old (`{ status: 'pending', created_at: { $lt: \u2026 }, run_id: null }`, sorted by\n `created_at`, 20 per tick; `status` and `created_at` are index slots, so the read stays cheap at\n any size) and starts each one with the same descriptor and the **same idempotency key** as\n `request-render`, so a row the app is re-driving at the same moment still gets one run. It\n **stops at the first `429`** (or `503 payments_unavailable`), since the rest of the batch would\n get the same answer, and the next tick carries on. Each try is recorded on the row (`redrive_attempts`, `redriven_at`). A `402`\n fails the row with `insufficient_credits`. A `400`, `401`, `403` or `422` also **stops** the tick\n and fails nothing: every re-driven row sends the same descriptor, so a refusal means the setup is\n wrong (a non-https or private `render_url`, a revoked key), not the row; the tick reports it as\n `stopped: { status, code }`. The only thing that fails a waiting row is age: a row still waiting\n after **one hour** is failed with `not_started: the render waited too long for a free slot`.\n Nothing was ever held, so nothing is refunded. In both the `402` and the one-hour case the owner\n is told once (`Idempotency-Key: render-not-started:<item_id>`), because the last thing they heard\n was \"queued\". (The function's scopes include `notifications:send` for that.)\n2. **The app**, calling `request-render` again with the same `request_key`, if it wants the render\n started sooner than the next tick.\n\n`overlap: 'skip'` keeps a slow tick from being doubled by the next one. On the Free plan, run it\nevery 15 minutes (see the plan note at the top).\n\n**Why it has its own key.** A credit hold placed by a function must name the signed-in user the\nfunction acts for, and a cron tick has none. So `redrive-pending` makes the start with\n`vxil_jobs_key`, an API key holding only `jobs:write`, the way your trusted server would. That is\nmore than \"hold credits and start runs\": `jobs:write` also lets the key cancel or replay any run of\nthis project, enqueue any job, and create or change schedules and flow rules, and it can hold\ncredits against any of your users and start runs against any public https endpoint. Treat it as a\nserver key: keep it only in this function's secrets, never in an app, and rotate it like any\nserver key. The re-driver never reads the\nprice from the row (a signed-in user can edit their own row): the hold is the function's fixed\n`RENDER_CREDITS`, and a row whose `render_key` is not `owner:request_key` is failed, not started.\n(A hand-written row like that which also breaks the `render_key_shape` hook cannot be written at\nall, so the tick counts it as `unwritable` and it keeps one of the 20 slots: fix or delete it by\nhand.)\n\n## A contract test for your runtime\n\nEvery key your runtime posts in the `completed` body is written onto the row, so every key must be a\ndeclared field of `renders`. Keep that true in your own CI with a test beside your runtime's code. It\nfails the day someone adds a key to the completion body and forgets the field:\n\n```ts\n// render-contract.test.ts \u2014 vitest, in YOUR repo (the one holding vxil.config.ts)\nimport { describe, expect, it } from 'vitest';\nimport config from './vxil.config';\n// the bodies your runtime (or coordinator) really posts: import the builders from that code,\n// so the test follows it instead of a hand-copied list\nimport { completedBody, progressBody } from './runtime/callback-bodies';\n\ndescribe('render callbacks', () => {\n const declared = new Set(Object.keys(config.cms!.collections!.renders!.fields!));\n\n it('every key of the completion body is a declared renders field', () => {\n const body = completedBody({ objectId: 'obj_test', durationS: 1 });\n expect(Object.keys(body).filter((k) => !declared.has(k))).toEqual([]);\n });\n\n it('a progress ping carries only the keys the mirror keeps', () => {\n const ping = progressBody({ progress: 40, stage: 'encoding' });\n expect(Object.keys(ping).filter((k) => !['status', 'progress', 'stage', 'message'].includes(k))).toEqual([]);\n });\n});\n```\n\nThe platform also has a fallback for the day that test is missing. When the row refuses a completion\nmirror because of an undeclared key, vxil writes the status alone, so the row still says\n`completed`, and the run reports what it dropped (`GET /v1/jobs/runs/{run_id}` \u2192 `mirror_error`,\nwith `fields_dropped`). The app no longer hangs on `processing`, but the dropped keys are not on the\nrow. The test is what keeps them there.\n\n## The bounds to design against\n\n- **Deadline**: the run's timeout is clamped to `generation.maxTimeoutMs` \u2014 one hour at most. A render\n that can take longer should be split (the next step starts from `notify-ready`), or tracked on your own\n row without a platform-held reserve.\n- **In flight**: `generation.maxConcurrent` renders at once (20 here and by default; up to 200). Over\n it, `request-render` answers `429` and the row waits for `redrive-pending`, or for a retry with the\n same `request_key` ([the backlog](#the-backlog-maxconcurrent-429-and-the-re-driver)).\n- **Holds**: one render's reserve is clamped to `generation.maxReserveCredits` (50 here), and the sum of\n all open holds is capped by `generation.maxOutstandingReserveCredits` (5,000 here).\n- **Callback**: at most 256 KiB per post, JSON, every key a declared field of `renders`. About 120\n posts a minute per render plus a small allowance kept for the final post (both best-effort, per\n serving machine); progress at most every 5 s, and the final post retried until it lands\n ([callback limits](#callback-limits-progress-is-best-effort-the-final-post-must-arrive)).\n\n## Your token, and where it lives\n\n`render_url` and `render_token` are function secrets; `request-render` reads them at invoke time and\nputs them on the generation run, which vxil stores with the run (your project only) for as long as the\njobs retention keeps it. The token is **never returned by a read**: `GET /v1/jobs/runs/{run_id}` (and\nthe dashboard and MCP reads built on it) shows the provider header names with every value as\n`[redacted]`. Rotate it in both places (`vxil secrets set functions/render_token` and your endpoint's\nenv); runs already started keep the token they were started with.\n\n## Evidence\n\n- **Read in vxil's code**: the start request body (`provider.body` + `payload` + `callback_url`), the\n 20-second start bound, the retry ladder, the status mirror writing every completion key onto the row,\n the hold committed on `completed` and released on `failed` / the deadline, and `error` / `hint`\n becoming `error_class` / `error_hint` on `job.generation.failed`.\n- **Read in Trigger.dev's and Modal's documentation, not executed from vxil**: the trigger endpoint\n `POST https://api.trigger.dev/api/v1/tasks/{taskId}/trigger` with `{ payload, options }` and the\n `idempotencyKey` / `ttl` / `tags` / `machine` options; `task({ id, machine, maxDuration, retry, run,\n onFailure })` and `retry` options, `maxDuration` being CPU time per attempt with no `onFailure` when\n it is exceeded, `metadata.set`, and the `ffmpeg()` / `puppeteer()` build extensions; Trigger.dev\n Cloud's US-hosted operational data; Modal's `@modal.fastapi_endpoint` and `.spawn()`. Check each\n vendor's current docs before you ship.\n",
|
|
18564
18606
|
"functions": {
|
|
18565
18607
|
"notify-ready.ts": "// notify-ready.ts \u2014 job.generation.completed | failed \u2192 tell the render's owner\n// (a vxil function, webhook trigger on `job.generation.`).\n//\n// The status mirror has already written the row (status, and on completion\n// every key your runtime sent back). This function adds the two things a\n// mirror cannot: the failure cause on the row, and a message to the user.\n//\n// AT-LEAST-ONCE: an event can be delivered again. And because the trigger\n// declares `retry: { maxAttempts: 3 }`, a non-2xx answer from here goes back\n// to the jobs ladder for another attempt (without `retry` it would simply be\n// acknowledged). Every attempt carries the same event. The notification carries an\n// Idempotency-Key of one per RUN, so a redelivery sends nothing twice, and the\n// row patch writes the same value again.\n\nimport type { JobGenerationSettledEventPayload, WebhookFunctionEnvelope } from '@vxil/sdk';\n\ntype RenderRow = {\n owner?: string; composition?: string; output_url?: string; error?: string; redrive_attempts?: number;\n run_id?: string | null; status?: string;\n};\n\n/** Platform error classes settle with no hint; these are the ones a render\n * can end with, in words for the row (the class stays first, for code). */\nconst PLATFORM_HINTS: Record<string, string> = {\n GenerationExpired: 'the render did not finish before its deadline',\n RetryableHttp: 'the render endpoint kept failing to accept the render (retries exhausted)',\n NetworkError: 'the render endpoint could not be reached (retries exhausted)',\n};\n\nfunction patchError(base: string, H: Record<string, string>, id: string, error: string): Promise<Response> {\n return fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(id)}`, {\n method: 'PATCH', headers: H, body: JSON.stringify({ data: { error } }),\n });\n}\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as WebhookFunctionEnvelope<JobGenerationSettledEventPayload>;\n const d = env.payload?.data;\n // job.generation.queued carries no status; a truncated event has no fields\n if (!d || 'truncated' in d || !('status' in d) || !d.generation_id) return Response.json({ skipped: true });\n const cms = env.scoped_jwts?.cms;\n const notifications = env.scoped_jwts?.notifications;\n if (!cms || !notifications) return Response.json({ error: 'missing cms/notifications scope' }, { status: 403 });\n const base = env.vxil_base;\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n\n const rowRes = await fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(d.generation_id)}`, { headers: H });\n // another job's generation event (not a render) \u2014 nothing to do\n if (rowRes.status === 404) return Response.json({ skipped: 'not a render' });\n if (!rowRes.ok) return Response.json({ error: `render read: ${rowRes.status}` }, { status: 502 }); // another attempt (header note)\n const row = ((await rowRes.json()) as { data: { data: RenderRow } }).data.data;\n if (!row.owner) return Response.json({ skipped: 'no owner' });\n\n if (d.status === 'failed') {\n // BRANCH ON error_class \u2014 not every `failed` is a render that failed:\n // \u2022 PaymentsUnavailable: the credits could not be held when the run was\n // started, so it ended before it began (nothing held). The start\n // answered 503 and the row is still pending \u2014 redrive-pending starts\n // it on a later tick (a NEW run). Nothing to tell the owner.\n // ONE race to undo: a request-render that re-sent the same key while\n // that start was still waiting on payments was handed THIS run (202,\n // deduplicated) and linked it \u2014 and redrive-pending only picks rows\n // with no run_id. Unlink it (only while the row is still pending on\n // this very run), so the re-driver starts it.\n if (d.error_class === 'PaymentsUnavailable') {\n if (row.run_id === d.run_id && (row.status ?? 'pending') === 'pending') {\n const unlinked = await fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(d.generation_id)}`, {\n method: 'PATCH', headers: H,\n body: JSON.stringify({ data: { run_id: null }, if: { run_id: d.run_id, status: 'pending' } }),\n });\n // 409: the row moved on meanwhile (re-driven, failed) \u2014 nothing to undo\n if (!unlinked.ok && unlinked.status !== 409) {\n return Response.json({ error: `render patch: ${unlinked.status}` }, { status: 502 }); // another attempt (header note)\n }\n return Response.json({ skipped: 'not started: payments unavailable', unlinked: unlinked.ok });\n }\n return Response.json({ skipped: 'not started: payments unavailable' });\n }\n // \u2022 Cancelled: your own cancel (POST /v1/jobs/runs/{id}/cancel); the\n // credits were released. The row says so; the owner is not told \"try\n // again\" \u2014 tell them from where you cancelled, if at all.\n if (d.error_class === 'Cancelled') {\n if (row.error !== 'cancelled') {\n const patched = await patchError(base, H, d.generation_id, 'cancelled');\n if (!patched.ok) return Response.json({ error: `render patch: ${patched.status}` }, { status: 502 }); // another attempt (header note)\n }\n return Response.json({ ok: true, notified: false });\n }\n // Not enough credits. A render request-render started itself already\n // answered the user (402) and wrote error: 'insufficient_credits' \u2014 keep\n // that code on the row (write it only if that write was lost) and send\n // nothing. A render the RE-DRIVER started is different: the user last\n // heard 429 \"queued\", so they are told \u2014 with the re-driver's own key\n // (render-not-started:<item_id>), so the two never both send.\n if (d.error_class === 'ReserveInsufficient') {\n if (!row.error) {\n const patched = await patchError(base, H, d.generation_id, 'insufficient_credits');\n if (!patched.ok) return Response.json({ error: `render patch: ${patched.status}` }, { status: 502 }); // another attempt (header note)\n }\n if (!(typeof row.redrive_attempts === 'number' && row.redrive_attempts > 0)) {\n return Response.json({ ok: true, notified: false });\n }\n return send(base, notifications, `render-not-started:${d.generation_id}`, row.owner, {\n subject: 'Your render could not start',\n paragraph: `\"${row.composition ?? 'Your render'}\" was queued, but there were not enough credits when its turn came. Nothing was charged \u2014 top up and try again.`,\n });\n }\n // the cause the run settled with: your runtime's `error` (+ `hint`), or\n // the platform's own class \u2014 which carries no hint, so a short human\n // one is added for the ones a user can meet (PLATFORM_HINTS)\n const hint = d.error_hint ?? (d.error_class ? PLATFORM_HINTS[d.error_class] : undefined);\n const cause = [d.error_class, hint].filter(Boolean).join(': ') || 'render failed';\n if (row.error !== cause) {\n const patched = await patchError(base, H, d.generation_id, cause.slice(0, 500));\n if (!patched.ok) return Response.json({ error: `render patch: ${patched.status}` }, { status: 502 }); // another attempt (header note)\n }\n }\n\n return send(base, notifications, `render-ready:${d.run_id}`, row.owner, d.status === 'completed'\n ? { subject: 'Your render is ready', paragraph: `\"${row.composition ?? 'Your render'}\" finished. Open the app to watch it.` }\n : { subject: 'Your render could not finish', paragraph: 'Nothing was charged \u2014 the credits are back on your balance. Try again in a minute.' });\n },\n};\n\n/** One transactional message; a non-2xx answer asks the ladder for another attempt. */\nasync function send(\n base: string, token: string, idempotencyKey: string, userId: string,\n data: { subject: string; paragraph: string },\n): Promise<Response> {\n const sent = await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json', 'idempotency-key': idempotencyKey },\n body: JSON.stringify({ user_id: userId, template: 'transactional', data }),\n });\n return sent.ok\n ? Response.json({ ok: true, notified: true })\n : Response.json({ error: `notifications send: ${sent.status}` }, { status: 502 }); // another attempt (header note)\n}\n",
|
|
18566
18608
|
"redrive-pending.ts": "// redrive-pending.ts \u2014 the BACKLOG RE-DRIVER (a vxil function, cron trigger,\n// every minute; `overlap: 'skip'`).\n//\n// cron-walk: drains-filter \u2014 every row it starts is PATCHed with its run_id (and\n// every row it gives up on to status 'failed'), which takes it out of the\n// `{ status: 'pending', run_id: null }` read; a 429 stops the tick early.\n//\n// Why it exists: at the generation concurrency cap (`generation.maxConcurrent`\n// open runs \u2014 20 in this blueprint, up to 200) request-render answers 429 and\n// leaves the row `pending` with NO run. Without this function only the caller\n// re-drives it (the same request_key again). With it, a backlog drains on its\n// own: each tick reads the oldest pending rows that have had no run for at\n// least REDRIVE_AFTER_MS, and starts each one with the SAME descriptor and the\n// SAME idempotency key (the row's render_key) request-render uses \u2014 so a row a\n// user is re-driving at the same moment still gets exactly one run.\n//\n// per row, by the jobs answer:\n// 202 (new run, or an open one handed back) \u2192 PATCH run_id (the status\n// mirror settles the row from here)\n// 202 deduplicated, generation_status 'failed' \u2192 PATCH run_id + status 'failed'\n// 402 (too few credits) \u2192 PATCH status 'failed',\n// error 'insufficient_credits',\n// and TELL the owner (they last\n// heard \"queued\", not \"refused\")\n// 429 (cap reached / too many credits held), \u2192 STOP the tick; the rest wait\n// or 503 payments_unavailable (credits could for the next one (the wait\n// not be held: nothing started or held) header is reported; the same\n// key starts a fresh run then)\n// 400 / 401 / 403 / 422 \u2192 STOP the tick, report it. Every\n// re-driven row sends the same\n// descriptor apart from its own\n// (already bounded) composition and\n// props, so a refusal is a SETUP\n// error \u2014 a non-https or private\n// render_url, a wrong or revoked\n// key, a config change \u2014 that\n// would fail every row the same\n// way. Nothing is failed for it:\n// fix the setup and the next tick\n// carries on.\n// 5xx / network \u2192 record the attempt, go on\n// A row still pending with no run after BACKLOG_MAX_AGE_MS is the ONLY thing\n// the re-driver gives up on: status 'failed', error 'not_started: \u2026' (nothing\n// was ever held for it), and the owner is told once\n// (Idempotency-Key render-not-started:<item_id>).\n// Every try is recorded on the row: redrive_attempts, redriven_at.\n//\n// A HAND-WRITTEN ROW that breaks render_key_shape (written before the hook\n// existed, or by a path that bypassed it) cannot be patched either \u2014 the\n// hook judges the merged row \u2014 so it would stay pending and take one of the\n// tick's slots for good. Fix or delete such rows by hand; the tick reports\n// them as `unwritable`.\n\n// THE KEY: a function-originated credit hold must name the user it acts for,\n// and a cron tick has no signed-in user \u2014 so the jobs call is made with\n// `vxil_jobs_key`, an API key of this same backend holding ONLY `jobs:write`\n// (a trusted server key may hold credits for any of your users; README \"The\n// backlog\"). The rows are read and written with the function's own scoped cms\n// token. The credits held are this blueprint's fixed price, never the row's\n// `credits` field \u2014 a signed-in user can edit their own row, so nothing the\n// re-driver charges or starts is read from a value they could have lowered.\n// The owner is told with the function's scoped notifications token.\n\nimport type { CronFunctionEnvelope } from '@vxil/sdk';\n\n/** What one render costs \u2014 the SAME number as request-render's RENDER_CREDITS. */\nconst RENDER_CREDITS = 5;\n/** The deadline \u2014 the SAME number as request-render's RENDER_DEADLINE_MS. */\nconst RENDER_DEADLINE_MS = 1_800_000;\n/** A row is the re-driver's only once request-render has had time to start it. */\nconst REDRIVE_AFTER_MS = 30_000;\n/** Rows started per tick (bounded: a tick is one function invocation). */\nconst MAX_REDRIVE_PER_TICK = 20;\n/** A row still waiting for a run after this long is failed (nothing is held). */\nconst BACKLOG_MAX_AGE_MS = 3_600_000;\n\ntype Env = CronFunctionEnvelope;\ntype RenderRow = {\n item_id: string;\n created_at?: string;\n data: {\n render_key?: string; request_key?: string; owner?: string; composition?: string;\n props?: Record<string, unknown>; created_at?: string; redrive_attempts?: number;\n };\n};\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n const notifications = env.scoped_jwts?.notifications;\n const jobsKey = env.secrets?.vxil_jobs_key;\n const renderUrl = env.secrets?.render_url;\n const renderToken = env.secrets?.render_token;\n if (!cms || !notifications) return Response.json({ error: 'missing cms/notifications scope' }, { status: 403 });\n if (!jobsKey || !renderUrl || !renderToken) {\n return Response.json(\n { error: 'store the secrets: vxil secrets set functions/vxil_jobs_key (an API key holding only jobs:write), functions/render_url, functions/render_token' },\n { status: 503 },\n );\n }\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n const now = Date.now();\n\n // the oldest pending rows with no run: `status` (s4) and `created_at` (t1)\n // are index slots, so the slots pick the rows and the `run_id: null` test\n // only runs over what they picked\n const filter = encodeURIComponent(JSON.stringify({\n status: 'pending',\n created_at: { $lt: new Date(now - REDRIVE_AFTER_MS).toISOString() },\n run_id: null,\n }));\n const listed = await fetch(\n `${base}/v1/cms/items/renders?filter=${filter}&sort=created_at&limit=${MAX_REDRIVE_PER_TICK}`,\n { headers: H },\n );\n if (!listed.ok) return Response.json({ error: `renders read: ${listed.status}` }, { status: 502 });\n const rows = ((await listed.json()) as { data?: { items?: RenderRow[] } }).data?.items ?? [];\n\n const out = { scanned: rows.length, started: 0, failed: 0, retry_later: 0, given_up: 0, unwritable: 0,\n notified: 0, notify_failed: 0,\n stopped: null as null | { status: number; code: string | null; retry_after: string | null } };\n const tell = async (row: RenderRow, why: 'credits' | 'waited') => {\n if (await notifyNotStarted(base, notifications, row, why)) out.notified += 1;\n else out.notify_failed += 1;\n };\n\n for (const row of rows) {\n const d = row.data;\n const attempts = (typeof d.redrive_attempts === 'number' ? d.redrive_attempts : 0) + 1;\n const createdAt = Date.parse(d.created_at ?? row.created_at ?? '');\n const mark = { redrive_attempts: attempts, redriven_at: new Date(now).toISOString() };\n\n // a row that cannot be a render request-render made (a hand-written row\n // missing its key parts), or one that has waited too long: fail it \u2014 no\n // run exists, so nothing is held and nothing is refunded\n const shapeOk = !!d.owner && !!d.request_key && !!d.composition\n && d.render_key === `${d.owner}:${d.request_key}`\n && d.composition.length <= 120 && JSON.stringify(d.props ?? {}).length <= 16_384;\n if (!shapeOk || (Number.isFinite(createdAt) && now - createdAt > BACKLOG_MAX_AGE_MS)) {\n const error = shapeOk\n ? 'not_started: the render waited too long for a free slot'\n : 'not_started: the row is not a render request-render created';\n // only while it is STILL pending with no run (a racing start wins)\n if (await patchRow(base, H, row.item_id, { ...mark, status: 'failed', error }, { status: 'pending', run_id: null })) {\n out.given_up += 1;\n // the owner last heard \"queued\" (a 429): say it will not happen. A\n // malformed row's owner is not trusted, so it is only failed.\n if (shapeOk) await tell(row, 'waited');\n } else if (!shapeOk) {\n out.unwritable += 1; // header note: fix it by hand\n }\n continue;\n }\n\n const enq = await startRun(base, jobsKey, renderUrl, renderToken, {\n itemId: row.item_id, user: d.owner!, renderKey: d.render_key!, requestKey: d.request_key!,\n composition: d.composition!, props: d.props ?? {},\n });\n if (!enq) { // network: try again next tick\n await patchRow(base, H, row.item_id, mark);\n out.retry_later += 1;\n continue;\n }\n if (enq.status === 402) {\n if (await patchRow(base, H, row.item_id, { ...mark, status: 'failed', error: 'insufficient_credits' })) await tell(row, 'credits');\n out.failed += 1;\n continue;\n }\n const code = enq.status === 429 || (enq.status >= 400 && enq.status !== 402)\n ? ((await enq.json().catch(() => ({}))) as { error?: { code?: string } }).error?.code ?? null\n : null;\n if (enq.status === 429 || (enq.status >= 400 && enq.status < 500) || code === 'payments_unavailable') {\n // 429: at the cap \u2014 the rest of the batch would get the same answer.\n // 503 payments_unavailable: the credits could not be held right now\n // (nothing was started or held) \u2014 the rest would get the same answer;\n // the next tick re-sends the SAME key and gets a fresh run.\n // 400 / 401 / 403 / 422: a setup error (header note) \u2014 failing rows for\n // it would be wrong and could not be undone. Either way: record the\n // try, stop, and let the next tick (Retry-After: seconds) go on.\n await patchRow(base, H, row.item_id, mark);\n out.stopped = { status: enq.status, code: enq.status === 429 ? null : code, retry_after: enq.headers.get('retry-after') };\n break;\n }\n if (!enq.ok) { // 5xx: try again next tick\n await patchRow(base, H, row.item_id, mark);\n out.retry_later += 1;\n continue;\n }\n const run = ((await enq.json()) as { data: { run_id: string; generation_status?: string; deduplicated?: boolean } }).data;\n if (run.deduplicated && run.generation_status === 'failed') {\n // the run already ENDED (its row write was lost): record it as failed\n await patchRow(base, H, row.item_id, { ...mark, run_id: run.run_id, status: 'failed' });\n out.failed += 1;\n continue;\n }\n await patchRow(base, H, row.item_id, { ...mark, run_id: run.run_id });\n out.started += 1;\n }\n // a 2xx either way: a cron tick is never retried (the next tick is the retry)\n return Response.json(out);\n },\n};\n\ninterface Start {\n itemId: string; user: string; renderKey: string; requestKey: string;\n composition: string; props: Record<string, unknown>;\n}\n\n/** The SAME webhook-mode generation request-render starts (the CI gate keeps\n * the two descriptors equal), sent with the jobs:write server key. Null on a\n * network error. */\nasync function startRun(base: string, key: string, renderUrl: string, renderToken: string, a: Start): Promise<Response | null> {\n return fetch(`${base}/v1/jobs/generation`, {\n // tenant-key: jobs:write via secret:vxil_jobs_key \u2014 called with the server\n // key, not the function's scoped token (header note); the CI gate checks\n // the secret is declared instead of a jobs scope\n method: 'POST',\n headers: { authorization: `Bearer ${key}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n job_name: 'render',\n provider: {\n url: renderUrl,\n method: 'POST',\n headers: { authorization: `Bearer ${renderToken}` },\n body: { render_id: a.itemId, user_id: a.user, composition: a.composition, props: a.props },\n },\n completion: { mode: 'webhook', status_path: 'status' },\n status_mirror: {\n feature: 'cms', collection: 'renders', record_id: a.itemId, column: 'status',\n progress_fields: ['progress', 'stage', 'message'],\n },\n reserve_credits: { amount: RENDER_CREDITS, user_id: a.user, credit_type: 'render_credits', reason: `render ${a.composition}` },\n timeout: { after_ms: RENDER_DEADLINE_MS },\n payload: {\n generation_id: a.itemId, correlation_id: a.requestKey,\n deadline_at: new Date(Date.now() + RENDER_DEADLINE_MS).toISOString(),\n },\n idempotency_key: a.renderKey,\n }),\n }).catch(() => null);\n}\n\n/** PATCH a render row (optionally only while `when` still holds); true when it landed. */\nasync function patchRow(\n base: string, H: Record<string, string>, itemId: string,\n data: Record<string, unknown>, when?: Record<string, unknown>,\n): Promise<boolean> {\n const res = await fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(itemId)}`, {\n method: 'PATCH',\n headers: H,\n body: JSON.stringify(when ? { data, if: when } : { data }),\n }).catch(() => undefined);\n return res?.ok ?? false;\n}\n\n/** Tell the owner a render they were told was queued will not start. One\n * message per render (Idempotency-Key render-not-started:<item_id> \u2014 the SAME\n * key notify-ready uses for a re-driven credit refusal, so the two never both\n * send). True when it was accepted. */\nasync function notifyNotStarted(base: string, token: string, row: RenderRow, why: 'credits' | 'waited'): Promise<boolean> {\n const name = row.data.composition ?? 'Your render';\n const res = await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: {\n authorization: `Bearer ${token}`, 'content-type': 'application/json',\n 'idempotency-key': `render-not-started:${row.item_id}`,\n },\n body: JSON.stringify({\n user_id: row.data.owner,\n template: 'transactional',\n data: why === 'credits'\n ? { subject: 'Your render could not start', paragraph: `\"${name}\" was queued, but there were not enough credits when its turn came. Nothing was charged \u2014 top up and try again.` }\n : { subject: 'Your render could not start', paragraph: `\"${name}\" waited over an hour for a free slot and was cancelled. Nothing was charged \u2014 try again later.` },\n }),\n }).catch(() => undefined);\n return res?.ok ?? false;\n}\n",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@vxil/cli",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.17.0",
|
|
4
4
|
"private": false,
|
|
5
5
|
"description": "The vxil CLI — init, quickstart, push, gen, secrets, keys, migrate, doctor, diff, functions, payments; installs the `vxil` command (npm i -g @vxil/cli).",
|
|
6
6
|
"license": "MIT",
|
|
@@ -46,9 +46,9 @@
|
|
|
46
46
|
},
|
|
47
47
|
"devDependencies": {
|
|
48
48
|
"miniflare": "^4.20260611.0",
|
|
49
|
-
"@vxil/config": "0.
|
|
50
|
-
"@vxil/feature-configs": "0.9.1",
|
|
49
|
+
"@vxil/config": "0.11.0",
|
|
51
50
|
"@vxil/runtime": "0.0.1",
|
|
51
|
+
"@vxil/feature-configs": "0.9.1",
|
|
52
52
|
"@vxil/types": "0.0.1"
|
|
53
53
|
},
|
|
54
54
|
"scripts": {
|