@vxil/cli 0.15.0 → 0.16.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/vxil.js CHANGED
@@ -3446,11 +3446,16 @@ var FilesConfigSchema = Type.Object({
3446
3446
  {
3447
3447
  // maximum caps are a defense-in-depth ceiling on tenant-editable storage —
3448
3448
  // object storage is cheap but the database-resident metadata + abuse aren't
3449
- // (pricing review 2026-07-10). 5 GiB/object, 1 TiB/tenant absolute; true PER-TIER clamps
3450
- // (Free/Dev 10 GB · Team 50 GB · Business 256 GB) are a follow-up needing the
3451
- // tenant tier threaded to files-v1 (plan tiers).
3449
+ // (pricing review 2026-07-10). 5 GiB/object, 1 TiB/tenant absolute.
3452
3450
  maxObjectBytes: Type.Integer({ default: 100 * 1024 * 1024, minimum: 1, maximum: 5 * 1024 * 1024 * 1024 }),
3453
- maxTotalBytes: Type.Integer({ default: 10 * 1024 * 1024 * 1024, minimum: 1, maximum: 1024 * 1024 * 1024 * 1024 }),
3451
+ // NO default (2026-10-04): UNSET means "the plan's storage ceiling"
3452
+ // (control-plane plans.ts FILES_MAX_TOTAL_BYTES_BY_TIER — Free/Developer
3453
+ // 10 GiB, Team 50 GiB, Business 256 GiB, Enterprise 1 TiB), resolved by
3454
+ // files-v1 where the quota is enforced, so a plan change applies at once.
3455
+ // A value is the project's own LOWER cap (above the plan → 422 at config
3456
+ // write; a downgrade clamps it). It used to default to 10 GiB on every
3457
+ // plan, so a Team project that never set it stayed at 10 GiB.
3458
+ maxTotalBytes: Type.Optional(Type.Integer({ minimum: 1, maximum: 1024 * 1024 * 1024 * 1024 })),
3454
3459
  maxObjectCount: Type.Integer({ default: 1e5, minimum: 1, maximum: 1e8 })
3455
3460
  },
3456
3461
  { default: {} }
@@ -3477,8 +3482,8 @@ var FilesConfigSchema = Type.Object({
3477
3482
  enabled: Type.Boolean({ default: false }),
3478
3483
  // per-bucket default; omitted = never expire by default
3479
3484
  defaultExpiresInSeconds: Type.Optional(Type.Integer({ minimum: 60 })),
3480
- sweepCron: Type.String({ default: "0 * * * *" })
3481
- // jobs-v1 TTL sweep schedule
3485
+ sweepCron: Type.String({ default: "*/15 * * * *" })
3486
+ // jobs-v1 TTL sweep schedule (every 15 min since 2026-10-04; was hourly)
3482
3487
  })),
3483
3488
  extractText: Type.Optional(Type.Object({
3484
3489
  enabled: Type.Boolean({ default: false }),
@@ -4538,9 +4543,9 @@ function lowerAllTriggerBindings(def, name) {
4538
4543
  return bindings;
4539
4544
  }
4540
4545
  var CONFIG_FILENAMES = ["vxil.config.ts", "vxil.config.mjs", "vxil.config.js"];
4541
- var VXIL_CONFIG_PKG_VERSION = "0.10.0";
4542
- var VXIL_SDK_PKG_VERSION = "0.15.0";
4543
- var VXIL_CLI_PKG_VERSION = "0.15.0";
4546
+ var VXIL_CONFIG_PKG_VERSION = "0.10.1";
4547
+ var VXIL_SDK_PKG_VERSION = "0.16.1";
4548
+ var VXIL_CLI_PKG_VERSION = "0.16.1";
4544
4549
  function ensureScaffoldPackageJson(cwd, opts = {}) {
4545
4550
  const file = resolve(cwd, "package.json");
4546
4551
  const wanted = {
@@ -8881,6 +8886,8 @@ var API_CHEATSHEET = [
8881
8886
  { feature: "jobs", what: "replay / cancel a run", cmd: "vxil api POST /v1/jobs/runs/<run_id>/replay \xB7 vxil api POST /v1/jobs/runs/<run_id>/cancel" },
8882
8887
  { feature: "jobs", what: "list schedules: held (fired, not started 2+ intervals), skipped_fires, overlap \u2014 fn-cron:* rows are function crons", cmd: "vxil api GET /v1/jobs/schedules" },
8883
8888
  { feature: "jobs", what: "create / pause / resume / delete a schedule", cmd: `vxil api POST /v1/jobs/schedules '{"job_name":"digest","target_url":"https://\u2026","cron":"0 8 * * *"}' \xB7 POST /v1/jobs/schedules/<schedule_id>/pause \xB7 POST \u2026/resume \xB7 vxil api DELETE /v1/jobs/schedules/<schedule_id>` },
8889
+ { feature: "jobs", what: "edit a schedule in place (cron, timezone, target, payload)", cmd: `vxil api PATCH /v1/jobs/schedules/<schedule_id> '{"cron":"0 9 * * *","timezone":"Europe/Amman"}'` },
8890
+ { feature: "jobs", what: "read a fan-in batch (counts by state, completed_at)", cmd: "vxil api GET /v1/jobs/batches/<batch_id>" },
8884
8891
  { feature: "jobs", what: "flow rules (rate + parallelism per endpoint or job)", cmd: `vxil api POST /v1/jobs/flow-rules '{"match_kind":"job_name","match_value":"resize","max_parallel":2}' \xB7 vxil api DELETE /v1/jobs/flow-rules/<rule_id>` },
8885
8892
  // files
8886
8893
  { feature: "files", what: "upload a local file (mint \u2192 PUT the bytes \u2192 complete)", cmd: "vxil files put <path> --user <user_id> [--content-type <t>]" },
@@ -11752,7 +11759,7 @@ var TOOLS = [
11752
11759
  {
11753
11760
  name: "jobs_enqueue_generation",
11754
11761
  feature: "jobs",
11755
- description: "Start a long-running EXTERNAL generation (a render, a model job): vxil calls provider.url (https; your provider key in provider.headers), then learns completion by POLLING completion.poll.url every interval_ms or by the provider's WEBHOOK (completion.mode 'webhook' \u2014 a signed single-use callback URL is appended as completion.callback.query_param). completion.status_path names the status field; status_map maps provider words to completed / failed / processing. status_mirror writes the status onto your record (e.g. cms collection + record_id); timeout.after_ms ends it as failed; reserve_credits holds a user's credits, committed on completion and released on failure. Answers 202 { run_id, generation_status: 'pending' } (deduplicated:true for a repeated idempotency_key); 429 with Retry-After when the project's open-generation cap is reached. Read it with jobs_get_run (generation_status, result, progress, mirror_error).",
11762
+ description: "Start a long-running EXTERNAL generation (a render, a model job): vxil calls provider.url (https; your provider key in provider.headers), then learns completion by POLLING completion.poll.url every interval_ms or by the provider's WEBHOOK (completion.mode 'webhook' \u2014 a signed single-use callback URL is appended as completion.callback.query_param). completion.status_path names the status field; status_map maps provider words to completed / failed / processing. status_mirror writes the status onto your record (e.g. cms collection + record_id); timeout.after_ms ends it as failed; reserve_credits holds a user's credits, committed on completion and released on failure; when payments cannot be reached the call answers 503 payments_unavailable (nothing started or held \u2014 retry the same call after Retry-After) unless reserve_credits.on_unavailable is 'proceed'. max_attempts = provider START calls (a 408/429/5xx retries; a provider Retry-After on 429/503 is honoured). Answers 202 { run_id, generation_status: 'pending' } (deduplicated:true for a repeated idempotency_key); 429 with Retry-After when the project's open-generation cap is reached. Read it with jobs_get_run (generation_status, result, progress, mirror_error).",
11756
11763
  inputSchema: {
11757
11764
  type: "object",
11758
11765
  properties: {
@@ -11807,7 +11814,8 @@ var TOOLS = [
11807
11814
  user_id: { type: "string" },
11808
11815
  credit_type: { type: "string" },
11809
11816
  credit_types: { type: "array", items: { type: "string" } },
11810
- reason: { type: "string" }
11817
+ reason: { type: "string" },
11818
+ on_unavailable: { type: "string", enum: ["fail", "proceed"] }
11811
11819
  },
11812
11820
  required: ["amount", "user_id"]
11813
11821
  },
@@ -11912,14 +11920,15 @@ var TOOLS = [
11912
11920
  {
11913
11921
  name: "files_create_upload_url",
11914
11922
  feature: "files",
11915
- description: "Mint a presigned PUT URL for a new file (bytes go straight to storage, never through Vxil). After uploading, call files_complete to make the object available.",
11923
+ description: "Mint a presigned PUT URL for a new file (bytes go straight to storage, never through Vxil). After uploading, call files_complete to make the object available. Optional expiresInSeconds (60 s to 100 years) makes the object expire and be deleted that long after the mint; the answer's expires_at is when (null = never), expires_in is the upload URL's life in seconds.",
11916
11924
  inputSchema: {
11917
11925
  type: "object",
11918
11926
  properties: {
11919
11927
  user_id: { type: "string", description: "Owning end_user_id (your user id)." },
11920
11928
  filename: { type: "string" },
11921
11929
  content_type: { type: "string" },
11922
- size_bytes: { type: "number" }
11930
+ size_bytes: { type: "number" },
11931
+ expiresInSeconds: { type: ["integer", "null"], minimum: 60, maximum: 31536e5, description: "Delete the object this many seconds after the mint, 60..3153600000 (100 years) (null = never; omitted = the files.ttl default when enabled, else never)." }
11923
11932
  },
11924
11933
  required: ["user_id", "filename", "content_type", "size_bytes"]
11925
11934
  },
@@ -17994,7 +18003,7 @@ export default defineConfig({
17994
18003
  "fal_key"
17995
18004
  ],
17996
18005
  "configSrc": "import { defineConfig } from '@vxil/config';\n\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n// \"AI Media Studio\" \u2014 images and video from fal.ai, with credits that cannot\n// leak. One function starts a render; vxil does the rest:\n//\n// start-render (your function, end-user mode)\n// \u2192 writes a `renders` row the user owns\n// \u2192 POST /v1/jobs/generation: a fal QUEUE job, credits HELD for it\n// vxil's generation lane\n// \u2192 calls fal with your key, the signed callback in `?fal_webhook=`\n// \u2192 fal POSTs back `{ status: 'OK' | 'ERROR', payload }`\n// \u2192 status_map reads OK/ERROR, result_path picks the images / the video\n// \u2192 the `renders` row gets generation_status + result\n// \u2192 credits COMMIT on success, REFUND on failure or timeout\n//\n// No fal adapter exists in vxil and none is needed: the three generic options\n// (`completion.callback.query_param`, `completion.status_map`,\n// `completion.result_path`) describe any \"POST the job, we call your webhook\"\n// vendor. The platform owns the part a function cannot \u2014 the held credits,\n// the timeout and the refund.\n//\n// The credits half is a payments INTEGRATION on the deterministic `mock`\n// provider \u2014 no provider account needed to try it. \"Credits\" are usage units,\n// not money; vxil is never in the flow of funds.\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\nexport default defineConfig({\n env: 'staging',\n\n features: {\n jobs: {\n enabled: true,\n generation: {\n maxConcurrent: 10,\n // fal images finish in seconds, an 8 s Veo clip in a few minutes \u2014\n // a run fal never calls back about fails (and refunds) at 15 minutes\n defaultTimeoutMs: 900_000,\n maxTimeoutMs: 1_800_000,\n pollMaxAttempts: 30,\n // a video costs 10 credits; one run may never hold more than 20\n maxReserveCredits: 20,\n // \u2026nor may all of this project's in-flight renders together hold more than 2,000\n maxOutstandingReserveCredits: 2_000,\n },\n },\n\n payments: {\n enabled: true,\n provider: 'mock',\n defaults: { currency: 'usd' },\n ledger: {\n productMap: {\n media_pack_100: { creditType: 'media_credits', amount: 100, period: 'once' },\n },\n // subscription tiers: a monthly top-up for subscribers\n tierMap: {\n studio: {\n entitlements: ['render'],\n quotas: {},\n rank: 10,\n grants: [{ creditType: 'media_credits', amount: 300, period: 'monthly' }],\n },\n },\n // a render that fails or times out gives its held credits back\n autoRefundOnJobFailure: true,\n },\n },\n\n cms: {\n // a render row is live the moment it is written\n draftPublish: false,\n // in end-user mode a collection with no owner is server-only \u2014 renders\n // declares one, so a signed-in user sees only their own renders\n strictEndUserScope: true,\n // Lane-A hook (guide ch. 7): the dedupe key IS owner + ':' + request_key,\n // server-enforced \u2014 so one user's request_key can never collide with, or\n // block, another user's (a body naming another owner is a 400 anyway).\n hooks: {\n render_dedupe_key: {\n collection: 'renders',\n event: 'beforeWrite',\n kind: 'validate',\n expr: \"item.dedupe_key == concat(item.owner, ':', item.request_key)\",\n message: \"dedupe_key must be owner + ':' + request_key\",\n },\n },\n },\n functions: { enabled: true },\n },\n\n cms: {\n collections: {\n renders: {\n singular: 'render',\n ownerField: 'owner',\n fields: {\n // THE DEDUPE ANCHOR \u2014 owner + ':' + the client's own request_key\n // (the hook above enforces the composition), so the key is per user.\n // A retried start (a double tap, a timeout) hits 409 instead of\n // paying twice, and start-render re-drives a start that never got\n // its run.\n dedupe_key: { type: 'string', required: true, unique: true, indexSlot: 's1' },\n request_key: { type: 'string', required: true },\n owner: { type: 'string', indexSlot: 's2' },\n kind: { type: 'string', required: true, indexSlot: 's3', validation: { enum: ['image', 'video'] } },\n // written by vxil's status mirror: pending \u2192 processing \u2192 completed | failed\n generation_status: { type: 'string', indexSlot: 's4' },\n credits: { type: 'int', indexSlot: 'n1' },\n created_at: { type: 'datetime', indexSlot: 't1' },\n prompt: { type: 'text', required: true },\n run_id: { type: 'text' },\n // written by vxil's status mirror on completion: the value at\n // `result_path` \u2014 fal's `images` list, or its `video` object\n result: { type: 'json' },\n },\n },\n },\n },\n\n functions: {\n // Starts ONE render for the signed-in user. Invoke it in END-USER mode (with\n // the user's session): the held credits are forced onto that user, and the\n // row is theirs.\n 'start-render': {\n entry: './functions/start-render.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write', 'jobs:write'],\n // your fal API key; it rides the provider call's Authorization header\n secrets: ['secret:fal_key'],\n // the function itself never calls fal \u2014 vxil's generation lane does\n egressAllow: [],\n // `vxil gen` types vx.fn['start-render'] from this\n signature: {\n input: { kind: 'string', prompt: 'string', request_key: 'string' },\n output:\n '{ item_id: string; run_id: string; credits: number }'\n + ' | { duplicate: true; request_key: string; item_id: string; run_id: string | null; generation_status: string }',\n },\n },\n },\n\n secrets: {\n fal_key: {\n feature: 'functions',\n description: 'your fal.ai API key (the \"Key \u2026\" credential) \u2014 vxil never holds a fal account of its own',\n },\n },\n});\n",
17997
- "readme": "# AI Media Studio \u2014 fal.ai images and Veo video, with credits that cannot leak\n\n```bash\nvxil init my-studio --template fal-media\ncd my-studio\nprintf '%s' \"$FAL_KEY\" | vxil secrets set functions/fal_key\nvxil push\n```\n\n> **Plan note.** The function deploys on the Free plan when the project's workload is `staging` or\n> `development` (`vxil projects workload <slug> development`, or create it with\n> `vxil projects create <slug> --workload development`). On a Free `production` project, `vxil push` stops before it writes anything, naming the plan and the ways out: change the workload or upgrade to Developer, or run `vxil push --skip-functions` to apply the collections and config without the functions.\n\nOne function starts a render; vxil carries it to the end. You bring a fal.ai key; vxil holds the\nuser's credits while fal works, writes the image or the video onto the user's `renders` row when\nfal calls back, and gives the credits back when fal fails or never answers.\n\n**What this blueprint teaches that the others do not:** a third-party *queue* API \u2014 \"POST the job,\nwe call your webhook\" \u2014 completed through the generation lane's **generic completion options**,\nwith no vendor adapter in vxil and no long call held open in a function:\n\n| option | what it does | fal's case |\n|---|---|---|\n| `completion.callback.query_param` | puts vxil's signed callback URL in the provider URL's query string, and sends `provider.body` exactly as written | fal reads its webhook from `?fal_webhook=` and the model input from the body |\n| `completion.status_map` | up to 8 of the provider's own status words \u2192 `completed` \xB7 `failed` \xB7 `processing` | fal's webhook says `OK` or `ERROR` \u2014 words vxil's built-in list would read as \"still working\" until the run timed out |\n| `completion.result_path` | the value the run settles with, written to your record as `result` | `payload.images` (a list of `{ url, width, height, \u2026 }`) or `payload.video` (`{ url, \u2026 }`) |\n\nThe same three options fit any vendor that calls you back (a transcode API, a render farm, a\nlong-running inference host) \u2014 change the URL, the words and the path.\n\n## What you get\n\n- **`renders`** \u2014 one row per request, owned by the user who started it (`strictEndUserScope`, so a\n signed-in user reads only their own renders). `dedupe_key` (the owner + `:` + the client's\n `request_key`, the composition enforced by a `beforeWrite` hook) is unique, so keys are **per user**: a\n double tap or a retried start is a `409` the function reads back. A row that already has its run is a\n duplicate; a row with no run (the first start died, or lost its enqueue answer, between the row and\n the enqueue) is **re-driven** \u2014 the same `dedupe_key` is the generation run's `idempotency_key`, so\n jobs hands back the existing run and fal is never asked twice.\n- **`start-render`** (http function, end-user mode) \u2014 writes the row, then enqueues ONE generation run:\n fal's queue URL, your key in `Authorization: Key \u2026`, the three options above, a status mirror onto\n the row, and `reserve_credits` for the render's cost (image 1, video 10 `media_credits`).\n- **credits** \u2014 a payments integration on the `mock` provider (no provider account needed). The hold is\n taken *before* the run is queued; a user without the credits gets `402` at once and nothing is held.\n Success commits the hold; a fal `ERROR`, a run fal never calls back about (the timeout: 5 min for an\n image, 15 for a video) or a cancel refunds it (`ledger.autoRefundOnJobFailure`).\n\n## Run it\n\nGive a user some credits from your server (or sell `media_pack_100` through your payments provider):\n\n```bash\ncurl -s -X POST \"https://api.vxil.com/v1/payments/credits/grant\" \\\n -H \"authorization: Bearer $KEY\" -H 'content-type: application/json' \\\n -H 'idempotency-key: welcome-u1' \\\n -d '{\"user_id\":\"<the user id>\",\"credit_type\":\"media_credits\",\"amount\":25,\"source\":\"welcome\"}'\n```\n\nStart a render **with the user's session** (end-user mode \u2014 the held credits are forced onto that\nuser, and a function cannot hold credits for anyone else):\n\n```ts\nimport { Vxil } from '@vxil/sdk';\n\n// after `vxil gen`, vx.fn['start-render'] is typed from the function's declared signature\nconst vx = new Vxil({ apiKey: process.env.VXIL_PUBLISHABLE_KEY!, endUserToken: process.env.USER_SESSION! });\nconst started = await vx.fn['start-render']({ kind: 'image', prompt: 'a red fox in the snow', request_key: 'fox-1' });\n// \u2192 { item_id, run_id, credits: 1 }\n// (or { duplicate: true, request_key, item_id, run_id, generation_status } on a retry)\n```\n\nThen read the row \u2014 or subscribe to its changes \u2014 until `generation_status` is `completed`:\n\n```ts\nif ('item_id' in started) {\n const row = await vx.from('renders').get(started.item_id);\n // row.generation_status \u2192 'completed'\n // row.result \u2192 [{ url: 'https://fal.media/files/\u2026png', width: 1024, height: 1024, \u2026 }]\n}\n```\n\nEvery run ends with `job.generation.completed` or `job.generation.failed` (`generation_id` = the row's\n`item_id`, `correlation_id` = its `request_key`), so a function bound to `job.generation.` can react \u2014\nsend a push, thumbnail the image \u2014 without a lookup table of its own.\n\n## How it fails, and what the user sees\n\n| what happened | the run | the row | the credits |\n|---|---|---|---|\n| fal answered `OK` | `completed` | `generation_status: completed`, `result` set | committed |\n| fal answered `ERROR` (its `error` text is the run's `last_error_msg`; `job.generation.failed` carries `error_class: 'ProviderFailed'`) | `failed` | `generation_status: failed` | refunded |\n| fal never called back | `failed` (`GenerationExpired`) at the timeout | `failed` | refunded |\n| the start call to fal failed with a 5xx / network fault | retried with backoff; terminal after the attempts | `processing` \u2192 the outcome above | held until then |\n| the user had too few credits | ended at once (`ReserveInsufficient`), never queued | `failed` (the function marks it); that `request_key` is spent \u2014 retry after a top-up with a new one | nothing held |\n| too many renders in flight, or a jobs-side fault | not created (the function answers the caller `429` with the wait in `retry_after`, or `502`) | `pending`, no run \u2014 calling start-render again with the **same** `request_key` re-drives it | nothing held |\n\n## Evidence\n\n- **Executed** (vxil's own integration tests):\n the `?fal_webhook=` callback URL is signed and verifies, the provider receives exactly the model input,\n `OK` settles completed with `payload.images` mirrored as `result` and the hold committed, `ERROR`\n settles failed with the hold refunded, a progress ping stays processing.\n- **Read in fal's documentation, not executed against fal from vxil** \u2014 the queue base\n `https://queue.fal.run/<model>`, the `fal_webhook` query parameter, the webhook body\n `{ request_id, status: 'OK' | 'ERROR', payload, error }`, and the model ids `fal-ai/flux/dev` and\n `fal-ai/veo3` with their inputs. Check the model page for each model's exact input fields before you\n ship, and keep `num_images` / `duration` to what your model accepts.\n\n## Your key, and where it lives\n\n`fal_key` is a function secret; `start-render` reads it at invoke time and puts it in the generation\nrun's provider headers, which vxil stores with the run (your project only) for as long as the jobs\nretention keeps the run, and sends to fal on the start call. It is **never returned by a read**:\n`GET /v1/jobs/runs/{run_id}` (and the dashboard and MCP reads built on it) shows the header names of a\ngeneration run with every value as `[redacted]`. Rotate it with `vxil secrets set functions/fal_key`;\nruns already queued keep the key they were started with.\n",
18006
+ "readme": "# AI Media Studio \u2014 fal.ai images and Veo video, with credits that cannot leak\n\n```bash\nmkdir my-studio && cd my-studio\nvxil init --template fal-media # init scaffolds into the CURRENT directory\nprintf '%s' \"$FAL_KEY\" | vxil secrets set functions/fal_key\nvxil push\n```\n\n> **Plan note.** The function deploys on the Free plan when the project's workload is `staging` or\n> `development` (`vxil projects workload <slug> development`, or create it with\n> `vxil projects create <slug> --workload development`). On a Free `production` project, `vxil push` stops before it writes anything, naming the plan and the ways out: change the workload or upgrade to Developer, or run `vxil push --skip-functions` to apply the collections and config without the functions.\n\nOne function starts a render; vxil carries it to the end. You bring a fal.ai key; vxil holds the\nuser's credits while fal works, writes the image or the video onto the user's `renders` row when\nfal calls back, and gives the credits back when fal fails or never answers.\n\n**What this blueprint teaches that the others do not:** a third-party *queue* API \u2014 \"POST the job,\nwe call your webhook\" \u2014 completed through the generation lane's **generic completion options**,\nwith no vendor adapter in vxil and no long call held open in a function:\n\n| option | what it does | fal's case |\n|---|---|---|\n| `completion.callback.query_param` | puts vxil's signed callback URL in the provider URL's query string, and sends `provider.body` exactly as written | fal reads its webhook from `?fal_webhook=` and the model input from the body |\n| `completion.status_map` | up to 8 of the provider's own status words \u2192 `completed` \xB7 `failed` \xB7 `processing` | fal's webhook says `OK` or `ERROR` \u2014 words vxil's built-in list would read as \"still working\" until the run timed out |\n| `completion.result_path` | the value the run settles with, written to your record as `result` | `payload.images` (a list of `{ url, width, height, \u2026 }`) or `payload.video` (`{ url, \u2026 }`) |\n\nThe same three options fit any vendor that calls you back (a transcode API, a render farm, a\nlong-running inference host) \u2014 change the URL, the words and the path.\n\n## What you get\n\n- **`renders`** \u2014 one row per request, owned by the user who started it (`strictEndUserScope`, so a\n signed-in user reads only their own renders). `dedupe_key` (the owner + `:` + the client's\n `request_key`, the composition enforced by a `beforeWrite` hook) is unique, so keys are **per user**: a\n double tap or a retried start is a `409` the function reads back. A row that already has its run is a\n duplicate; a row with no run (the first start died, or lost its enqueue answer, between the row and\n the enqueue) is **re-driven** \u2014 the same `dedupe_key` is the generation run's `idempotency_key`, so\n jobs hands back the existing run and fal is never asked twice.\n- **`start-render`** (http function, end-user mode) \u2014 writes the row, then enqueues ONE generation run:\n fal's queue URL, your key in `Authorization: Key \u2026`, the three options above, a status mirror onto\n the row, and `reserve_credits` for the render's cost (image 1, video 10 `media_credits`).\n- **credits** \u2014 a payments integration on the `mock` provider (no provider account needed). The hold is\n taken *before* the run is queued; a user without the credits gets `402` at once and nothing is held.\n Success commits the hold; a fal `ERROR`, a run fal never calls back about (the timeout: 5 min for an\n image, 15 for a video) or a cancel refunds it (`ledger.autoRefundOnJobFailure`).\n\n## Run it\n\nGive a user some credits from your server (or sell `media_pack_100` through your payments provider):\n\n```bash\ncurl -s -X POST \"https://api.vxil.com/v1/payments/credits/grant\" \\\n -H \"authorization: Bearer $KEY\" -H 'content-type: application/json' \\\n -H 'idempotency-key: welcome-u1' \\\n -d '{\"user_id\":\"<the user id>\",\"credit_type\":\"media_credits\",\"amount\":25,\"source\":\"welcome\"}'\n```\n\nStart a render **with the user's session** (end-user mode \u2014 the held credits are forced onto that\nuser, and a function cannot hold credits for anyone else):\n\n```ts\nimport { Vxil } from '@vxil/sdk';\n\n// after `vxil gen`, vx.fn['start-render'] is typed from the function's declared signature\nconst vx = new Vxil({ apiKey: process.env.VXIL_PUBLISHABLE_KEY!, endUserToken: process.env.USER_SESSION! });\nconst started = await vx.fn['start-render']({ kind: 'image', prompt: 'a red fox in the snow', request_key: 'fox-1' });\n// \u2192 { item_id, run_id, credits: 1 }\n// (or { duplicate: true, request_key, item_id, run_id, generation_status } on a retry)\n```\n\nThen read the row \u2014 or subscribe to its changes \u2014 until `generation_status` is `completed`:\n\n```ts\nif ('item_id' in started) {\n const row = await vx.from('renders').get(started.item_id);\n // row.generation_status \u2192 'completed'\n // row.result \u2192 [{ url: 'https://fal.media/files/\u2026png', width: 1024, height: 1024, \u2026 }]\n}\n```\n\nEvery run ends with `job.generation.completed` or `job.generation.failed` (`generation_id` = the row's\n`item_id`, `correlation_id` = its `request_key`), so a function bound to `job.generation.` can react \u2014\nsend a push, thumbnail the image \u2014 without a lookup table of its own.\n\n## How it fails, and what the user sees\n\n| what happened | the run | the row | the credits |\n|---|---|---|---|\n| fal answered `OK` | `completed` | `generation_status: completed`, `result` set | committed |\n| fal answered `ERROR` (its `error` text is the run's `last_error_msg`; `job.generation.failed` carries `error_class: 'ProviderFailed'`) | `failed` | `generation_status: failed` | refunded |\n| fal never called back | `failed` (`GenerationExpired`) at the timeout | `failed` | refunded |\n| the start call to fal failed with a 5xx / network fault | retried with backoff; terminal after the attempts | `processing` \u2192 the outcome above | held until then |\n| the user had too few credits | ended at once (`ReserveInsufficient`), never queued | `failed` (the function marks it); that `request_key` is spent \u2014 retry after a top-up with a new one | nothing held |\n| too many renders in flight, or a jobs-side fault | not created (the function answers the caller `429` with the wait in `retry_after`, or `502`) | `pending`, no run \u2014 calling start-render again with the **same** `request_key` re-drives it | nothing held |\n\n## Evidence\n\n- **Executed** (vxil's own integration tests):\n the `?fal_webhook=` callback URL is signed and verifies, the provider receives exactly the model input,\n `OK` settles completed with `payload.images` mirrored as `result` and the hold committed, `ERROR`\n settles failed with the hold refunded, a progress ping stays processing.\n- **Read in fal's documentation, not executed against fal from vxil** \u2014 the queue base\n `https://queue.fal.run/<model>`, the `fal_webhook` query parameter, the webhook body\n `{ request_id, status: 'OK' | 'ERROR', payload, error }`, and the model ids `fal-ai/flux/dev` and\n `fal-ai/veo3` with their inputs. Check the model page for each model's exact input fields before you\n ship, and keep `num_images` / `duration` to what your model accepts.\n\n## Your key, and where it lives\n\n`fal_key` is a function secret; `start-render` reads it at invoke time and puts it in the generation\nrun's provider headers, which vxil stores with the run (your project only) for as long as the jobs\nretention keeps the run, and sends to fal on the start call. It is **never returned by a read**:\n`GET /v1/jobs/runs/{run_id}` (and the dashboard and MCP reads built on it) shows the header names of a\ngeneration run with every value as `[redacted]`. Rotate it with `vxil secrets set functions/fal_key`;\nruns already queued keep the key they were started with.\n",
17998
18007
  "functions": {
17999
18008
  "start-render.ts": "// start-render.ts \u2014 start ONE fal.ai render for the signed-in user (a vxil\n// function, http trigger, END-USER mode).\n//\n// POST /v1/fn/start-render (with the user's session)\n// { \"kind\": \"image\" | \"video\", \"prompt\": \"\u2026\", \"request_key\": \"<your idempotency key>\" }\n// \u2192 202 { item_id, run_id, credits } a new render, credits held\n// \u2192 200 { duplicate: true, request_key, item_id, run_id, generation_status }\n// this user's request_key already started one\n//\n// What happens after the 202 is vxil's, not this function's:\n// \u2022 the generation lane POSTs fal's QUEUE API with your key and the signed\n// callback URL in `?fal_webhook=` (completion.callback.query_param);\n// \u2022 fal calls back `{ status: 'OK' | 'ERROR', payload, error? }` \u2014 status_map\n// turns its words into completed / failed;\n// \u2022 result_path picks `payload.images` (or `payload.video`), and the status\n// mirror writes it onto this render's `result` with generation_status;\n// \u2022 the held credits commit on completed and are REFUNDED on failed, on a\n// run fal never calls back about (the timeout), and on a cancel.\n// Every run ends with job.generation.completed or job.generation.failed\n// (generation_id = the render's item_id, correlation_id = its request_key) if\n// you want to react to it in another function.\n//\n// DELIVERY IS AT-LEAST-ONCE and users double-tap: the row's `dedupe_key`\n// (owner + ':' + request_key \u2014 PER USER, so one user's key can never block\n// another's) is declared unique, so a second start with the same key is a\n// 409. On a 409 we read THIS user's row: a row that already has its run is a\n// duplicate; a row with no run (the first start died, or its enqueue answer\n// was lost, between the row and the enqueue) is RE-DRIVEN \u2014 the generation\n// run carries the same dedupe_key as its idempotency_key, so jobs hands back\n// the existing run instead of enqueueing a second fal job.\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\ntype Input = { kind?: string; prompt?: string; request_key?: string };\ntype Env = HttpFunctionEnvelope<Input>;\ntype RenderRow = { item_id: string; data: { kind?: string; prompt?: string; run_id?: string; generation_status?: string } };\n\n/** fal model ids (https://fal.ai/models) \u2014 swap freely; both are queue APIs. */\nconst MODELS = {\n image: { url: 'https://queue.fal.run/fal-ai/flux/dev', credits: 1, resultPath: 'payload.images' },\n video: { url: 'https://queue.fal.run/fal-ai/veo3', credits: 10, resultPath: 'payload.video' },\n} as const;\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n const jobs = env.scoped_jwts?.jobs;\n const falKey = env.secrets?.fal_key;\n if (!cms || !jobs) return Response.json({ error: 'missing cms/jobs scope' }, { status: 403 });\n if (!falKey) return Response.json({ error: 'store your fal key: vxil secrets set functions/fal_key' }, { status: 500 });\n // the held credits are FORCED onto the verified end-user \u2014 a function\n // cannot hold credits against a user it does not act for\n const user = env.end_user?.id;\n if (!user) return Response.json({ error: 'invoke start-render with the user\\'s session (end-user mode)' }, { status: 401 });\n\n const kind = env.payload?.kind === 'video' ? 'video' : env.payload?.kind === 'image' ? 'image' : null;\n const prompt = typeof env.payload?.prompt === 'string' ? env.payload.prompt.trim().slice(0, 2_000) : '';\n const requestKey = typeof env.payload?.request_key === 'string' ? env.payload.request_key.slice(0, 120) : '';\n if (!kind || !prompt || !requestKey) {\n return Response.json({ error: 'need { kind: image|video, prompt, request_key }' }, { status: 422 });\n }\n const dedupeKey = `${user}:${requestKey}`;\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n\n // 1. the render row (owned by the user \u2014 cms forces `owner` in end-user mode;\n // the render_dedupe_key hook checks dedupe_key = owner + ':' + request_key)\n const created = await fetch(`${base}/v1/cms/items/renders`, {\n method: 'POST',\n headers: H,\n body: JSON.stringify({\n data: {\n dedupe_key: dedupeKey, request_key: requestKey, owner: user, kind, prompt,\n credits: MODELS[kind].credits, generation_status: 'pending', created_at: new Date().toISOString(),\n },\n }),\n });\n if (created.status === 409) {\n // THIS user's row for this key (the read is owner-scoped in end-user mode)\n const filter = encodeURIComponent(JSON.stringify({ dedupe_key: dedupeKey }));\n const found = await fetch(`${base}/v1/cms/items/renders?filter=${filter}&limit=1`, { headers: H });\n const row = found.ok ? ((await found.json()) as { data?: { items?: RenderRow[] } }).data?.items?.[0] : undefined;\n if (!row) return Response.json({ error: `render lookup: ${found.status}` }, { status: 502 });\n if (row.data.run_id || row.data.generation_status === 'failed') {\n return Response.json({\n duplicate: true, request_key: requestKey, item_id: row.item_id,\n run_id: row.data.run_id ?? null, generation_status: row.data.generation_status ?? 'pending',\n });\n }\n // a start that never got its run: re-drive it (jobs dedupes on dedupe_key)\n const rowKind = row.data.kind === 'video' ? 'video' : 'image';\n return startRun(base, H, jobs, falKey, user, dedupeKey, requestKey, row.item_id, rowKind, row.data.prompt ?? prompt);\n }\n if (!created.ok) return Response.json({ error: `render row: ${created.status}` }, { status: 502 });\n const itemId = ((await created.json()) as { data: { item_id: string } }).data.item_id;\n return startRun(base, H, jobs, falKey, user, dedupeKey, requestKey, itemId, kind, prompt);\n },\n};\n\n/** 2. the fal queue job, babysat by vxil, credits held for it. Idempotent on\n * `dedupeKey`: a re-drive gets the run that already exists. */\nasync function startRun(\n base: string, H: Record<string, string>, jobs: string, falKey: string, user: string,\n dedupeKey: string, requestKey: string, itemId: string, kind: 'image' | 'video', prompt: string,\n): Promise<Response> {\n const model = MODELS[kind];\n const enq = await fetch(`${base}/v1/jobs/generation`, {\n method: 'POST',\n headers: { authorization: `Bearer ${jobs}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n job_name: `fal.${kind}`,\n // the body is EXACTLY fal's model input (query-string callbacks send\n // provider.body as-is \u2014 no payload / callback_url keys merged in)\n provider: {\n url: model.url,\n method: 'POST',\n headers: { authorization: `Key ${falKey}` },\n body: kind === 'image' ? { prompt, num_images: 1 } : { prompt, duration: '8s' },\n },\n completion: {\n mode: 'webhook',\n status_path: 'status',\n callback: { query_param: 'fal_webhook' },\n status_map: { OK: 'completed', ERROR: 'failed' },\n result_path: model.resultPath,\n },\n status_mirror: { feature: 'cms', collection: 'renders', record_id: itemId, column: 'generation_status' },\n reserve_credits: { amount: model.credits, user_id: user, credit_type: 'media_credits', reason: `fal ${kind}` },\n timeout: { after_ms: kind === 'image' ? 300_000 : 900_000 },\n // rides job.generation.* as generation_id / correlation_id\n payload: { generation_id: itemId, correlation_id: requestKey },\n idempotency_key: dedupeKey,\n }),\n });\n if (enq.status === 402) {\n // not enough credits: the run already ENDED (job.generation.failed,\n // ReserveInsufficient) and nothing was held \u2014 mark the row and say so.\n // This key is spent; a new attempt (after a top-up) uses a new request_key.\n await markFailed(base, H, itemId);\n return Response.json({ error: 'insufficient_credits', item_id: itemId }, { status: 402 });\n }\n if (!enq.ok) {\n // 429 (too many in flight) / 5xx: NO run is promised \u2014 leave the row\n // pending with no run, so a retry with the SAME request_key re-drives it\n return Response.json(\n { error: `generation enqueue: ${enq.status}`, item_id: itemId, retry_after: enq.headers.get('retry-after') },\n { status: enq.status === 429 ? 429 : 502 },\n );\n }\n const runId = ((await enq.json()) as { data: { run_id: string } }).data.run_id;\n await fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(itemId)}`, {\n method: 'PATCH',\n headers: H,\n body: JSON.stringify({ data: { run_id: runId } }),\n });\n return Response.json({ item_id: itemId, run_id: runId, credits: model.credits }, { status: 202 });\n}\n\nasync function markFailed(base: string, H: Record<string, string>, itemId: string): Promise<void> {\n await fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(itemId)}`, {\n method: 'PATCH',\n headers: H,\n body: JSON.stringify({ data: { generation_status: 'failed' } }),\n });\n}\n"
18000
18009
  }
@@ -18250,6 +18259,39 @@ export default defineConfig({
18250
18259
  "archive-document.ts": "// archive-document.ts \u2014 the ARCHIVE BUTTON (a vxil function).\n//\n// Trigger: the per-record action `archive` declared on the `documents`\n// collection. Pressing the button POSTs\n// /v1/cms/items/documents/<id>/actions/archive\n// and the platform invokes THIS function once with the action envelope as its\n// payload:\n// { collection, item_id, action, actor, item: { item_id, status, version, data } }\n//\n// The pressing member's verified principal is carried into the scoped token, so\n// the PATCH below is owner-scoped exactly as if the member had written it \u2014 a\n// member can archive their own document and nobody else's, with no check here.\n//\n// It writes ONE transition (\u2192 'archived'). The collection's lifecycle hook is\n// still the authority: an illegal transition is rejected in the same write, and\n// this function reports that rejection instead of pretending it succeeded.\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\ninterface ActionPayload {\n collection?: string;\n item_id?: string;\n action?: string;\n actor?: { principal?: string; end_user_id?: string };\n item?: { status?: string; version?: number; data?: Record<string, unknown> };\n}\ntype Env = HttpFunctionEnvelope<ActionPayload>;\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n const itemId = env.payload?.item_id;\n // The platform now filters cms-hook deliveries on the binding's collection/event\n // server-side (guide ch. 8); this guard stays as belt-and-braces.\n if (!cms || !itemId || env.payload?.collection !== 'documents') {\n return Response.json({ skipped: true, reason: 'not a documents action' });\n }\n\n // Already archived \u2192 nothing to do. The action is human-initiated and can be\n // pressed twice; make the second press a no-op rather than an error.\n const was = String(env.payload?.item?.data?.state ?? '');\n if (was === 'archived') {\n return Response.json({ archived: itemId, already: true, state: 'archived' });\n }\n\n const res = await fetch(`${base}/v1/cms/items/documents/${itemId}`, {\n method: 'PATCH',\n headers: { authorization: `Bearer ${cms}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n data: { state: 'archived', updated_at: new Date().toISOString() },\n }),\n });\n\n if (!res.ok) {\n // The lifecycle hook refuses an illegal transition in-transaction (422).\n // Surface the real reason \u2014 the action route relays this class straight\n // back to the caller as `action_failed`.\n const detail = await res.text();\n return Response.json(\n { error: 'archive_refused', from: was, status: res.status, detail: detail.slice(0, 300) },\n { status: res.status === 422 ? 422 : 502 },\n );\n }\n\n return Response.json({ archived: itemId, from: was || 'draft', state: 'archived' });\n },\n};\n"
18251
18260
  }
18252
18261
  },
18262
+ {
18263
+ "id": "admin-console",
18264
+ "title": "Admin console (accounts \xB7 approvals \xB7 audit)",
18265
+ "vertical": "saas",
18266
+ "summary": "The back-office every internal-tools team builds first: staff sign in through your company identity provider, every read and write goes through a permission-checked function, look up and edit customer accounts, and run bulk fixes of up to 25 accounts at a time. Sensitive changes (usage-credit grants, plan overrides, suspensions, data fixes) become requests that someone else approves: self-approval is impossible by construction, small grants auto-approve under a threshold and a daily cap stored as a row (splitting a grant still reaches an approver), two approvers can never both apply, and stale requests expire. Each account shows its change history and internal notes, and two fields are visible only to the roles that need them. Approvals are rows plus one deciding function, not a workflow engine.",
18267
+ "collections": [
18268
+ "accounts",
18269
+ "change_requests",
18270
+ "approval_policies"
18271
+ ],
18272
+ "features": [
18273
+ "auth",
18274
+ "orgs",
18275
+ "cms",
18276
+ "comments",
18277
+ "notifications",
18278
+ "functions"
18279
+ ],
18280
+ "hasFunctions": true,
18281
+ "byoKeys": [
18282
+ "vxil_read_key",
18283
+ "oidc_client_secret"
18284
+ ],
18285
+ "configSrc": "import { defineConfig } from '@vxil/config';\n\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n// \"Admin console\" \u2014 the back-office every internal-tools team builds first.\n//\n// Staff look up and edit customer accounts and run bulk fixes. Sensitive\n// changes (usage-credit grants, plan overrides, suspensions, data fixes) become\n// REQUESTS that a different person approves. Each account shows its change\n// history and internal notes. Declared end-to-end in ONE typed file:\n//\n// \u2022 auth \u2192 staff only: sign-in through your company identity provider,\n// fenced to your e-mail domain(s); the org role rides the session\n// \u2022 orgs \u2192 ONE staff organization; three custom roles (support, approver,\n// console-admin) are rows you create over the API (README)\n// \u2022 cms \u2192 accounts \xB7 change_requests \xB7 approval_policies, with the\n// approval STATE MACHINE and \"no self-approval\" as write hooks\n// \u2022 comments \u2192 internal notes, one thread per account (`accounts:<id>`),\n// read and written only through the console-api function\n// \u2022 notifications \u2192 \"approval needed\" / \"approval decided\" messages\n// \u2022 functions \u2192 every read AND write a staff member makes goes through one\n// of these, and each one checks the caller's permission first\n// (the browser key holds only functions:invoke)\n//\n// What this is NOT: an approval engine or a visual flow. An approval is a ROW\n// with a state, a hook that guards its transitions, and ONE function that\n// decides it. A second approval step is a second state and a second\n// permission, in data \u2014 never a graph.\n//\n// Credits here are USAGE UNITS your product consumes (API calls, seats-days,\n// exports), never money. Refunds through your own payment provider are out of\n// scope; see the README for the payments-integration functions pattern.\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\nexport default defineConfig({\n env: 'staging',\n\n features: {\n auth: {\n // Staff sign in through the company identity provider ONLY: no\n // password, no magic link, so there is no staff password to phish, reset\n // or leak.\n methods: { emailPassword: false, magicLink: false },\n providers: {\n oidc: {\n issuer: 'https://login.example-idp.com', // your IdP's issuer (https, no query/fragment)\n clientId: 'vxil-admin-console', // not a secret \u2014 it rides every authorize URL\n clientSecretRef: 'secret:oidc_client_secret', // the `secrets` block below\n scopes: ['email', 'profile'], // `openid` is always added; add 'groups' if your IdP needs it for the claim\n // `groups` from the IdP rides the session as roles, next to the org role.\n // Keep IdP group names distinct from the console's role names: a group\n // called `console-admin` would pass `readRoles` below.\n claims: { email: 'email', name: 'name', roles: 'groups' },\n // fail-closed to your domain(s), but that is EVERYONE at the company,\n // not only staff. Assign the console's app in your IdP to the staff\n // group too; the functions refuse non-staff either way (403).\n allowedDomains: ['example.com'],\n autoLink: false, // link by the IdP's stable subject only\n },\n },\n // The member's org role (support / approver / console-admin) is embedded\n // in the session at sign-in and refresh, so field-level read gates\n // (`readRoles` below) work with no round-trip. The functions still ask\n // the orgs permission check on every write: that is revocation-grade.\n orgClaims: { enabled: true },\n session: { ttlMinutes: 60, refreshTtlDays: 7 },\n security: {\n // Inert while sign-in is IdP-only (it guards password and one-time-code\n // sign-in); kept so a break-glass password login, if you ever turn one\n // on, is rate-limited from the first attempt.\n lockout: { maxFailures: 5, windowMinutes: 15, lockMinutes: 30 },\n // EVERY return URL must be one of these: the console, and a local dev\n // server (plain http is accepted on localhost / 127.0.0.1 only).\n allowedRedirectOrigins: ['https://console.example.com', 'http://localhost:5173'],\n },\n },\n\n // ONE organization holds your staff. Roles are rows (see the README):\n // support \u2192 accounts.read, changes.request, notes.write\n // approver \u2192 accounts.read, changes.approve\n // console-admin \u2192 all of the above + audit.read, policies.manage\n // The org `owner` holds every permission (break-glass).\n orgs: { enabled: true },\n\n // Records are live data, not editorial content: no draft step.\n cms: {\n draftPublish: false,\n hooks: {\n // \u2500\u2500 The approval state machine \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n // pending \u2192 approved | rejected | expired ; approved \u2192 applied | failed.\n // Everything else is terminal. A decision is two writes inside ONE\n // transaction (pending \u2192 approved, then approved \u2192 applied), so the\n // hook sees every step and `approved` is never left behind on success.\n request_stage: {\n collection: 'change_requests',\n event: 'beforeUpdate',\n kind: 'validate',\n expr:\n 'item.state == before.state'\n + \" || (before.state == 'pending' && (item.state == 'approved' || item.state == 'rejected' || item.state == 'expired'))\"\n + \" || (before.state == 'approved' && (item.state == 'applied' || item.state == 'failed'))\",\n message: 'illegal change-request state transition',\n },\n // A request is born pending \u2014 or already applied when a policy\n // auto-approved it (then decided_by is 'policy' and auto is true).\n request_born: {\n collection: 'change_requests',\n event: 'beforeCreate',\n kind: 'validate',\n expr:\n \"item.state == 'pending'\"\n + \" || (item.state == 'applied' && item.auto == true && item.decided_by == 'policy')\",\n message: \"a change request starts 'pending' (or 'applied' by policy)\",\n },\n // \u2500\u2500 Self-approval is impossible BY CONSTRUCTION \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n // Write hooks see the ROW, not the caller, so the human ids are fields:\n // `requested_by` is set by request-change from the verified session,\n // `decided_by` by decide-change from the verified session. Even a buggy\n // function cannot move a request into approved/applied with the same\n // person on both sides.\n no_self_approval: {\n collection: 'change_requests',\n event: 'beforeUpdate',\n kind: 'validate',\n expr:\n 'item.state == before.state'\n + \" || (item.state != 'approved' && item.state != 'applied')\"\n + ' || (not(isNull(item.decided_by)) && item.decided_by != item.requested_by)',\n message: 'self_approval: the requester cannot approve their own change',\n },\n // Credits are usage units and never go negative. (`$inc` writes skip\n // hooks; the `validation.min` on the field guards those, as a 409.)\n credits_nonneg: {\n collection: 'accounts',\n event: 'beforeWrite',\n kind: 'validate',\n expr: 'coalesce(item.credits, 0) >= 0',\n message: 'credits cannot go below zero',\n },\n // active \u2194 suspended; anything \u2192 closed; closed is terminal.\n account_status: {\n collection: 'accounts',\n event: 'beforeUpdate',\n kind: 'validate',\n expr:\n 'item.status == before.status'\n + \" || item.status == 'closed'\"\n + \" || (before.status == 'active' && item.status == 'suspended')\"\n + \" || (before.status == 'suspended' && item.status == 'active')\",\n message: 'illegal account status transition',\n },\n },\n },\n\n // Internal notes: one thread per account, topic `accounts:<item_id>`.\n // Comments are author-owned (a signed-in caller can only write as itself).\n comments: {},\n\n // Approval messages go to the inbox and by e-mail. `mock` keeps the\n // walkthrough provider-free; switch to your e-mail provider for real mail.\n notifications: { provider: 'mock', fromEmail: 'console@example.com', inboxEnabled: true },\n\n functions: { enabled: true },\n },\n\n // \u2500\u2500 Schema-as-code (\u22648 index slots per collection: s1\u2013s4/n1\u2013n2/t1\u2013t2) \u2500\u2500\u2500\u2500\u2500\u2500\n cms: {\n collections: {\n accounts: {\n singular: 'account',\n fields: {\n name: { type: 'string', required: true, indexSlot: 's1' }, // $startsWith search\n email: { type: 'string', required: true, unique: true, indexSlot: 's2' },\n plan: { type: 'string', indexSlot: 's3' },\n status: { type: 'string', indexSlot: 's4', validation: { enum: ['active', 'suspended', 'closed'] } },\n credits: { type: 'int', indexSlot: 'n1', validation: { min: 0 } }, // usage units, never money\n seats: { type: 'int', indexSlot: 'n2', validation: { min: 0 } },\n created_at: { type: 'datetime', indexSlot: 't1' },\n last_active_at: { type: 'datetime', indexSlot: 't2' },\n // the id of this account in YOUR product database \u2014 equality lookups\n // are index-served without spending a slot\n external_ref: { type: 'string', indexed: true },\n notes_count: { type: 'int', validation: { min: 0 } },\n // \u2500\u2500 FIELD-LEVEL READ SECURITY \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n // A signed-in staff member receives these only when the session's\n // role is listed. Everyone else gets the account WITHOUT the field \u2014\n // absent, not null \u2014 and cannot filter or sort on it.\n risk_flags: { type: 'json', readRoles: ['console-admin', 'owner'] },\n billing_contact: { type: 'string', readRoles: ['console-admin', 'approver', 'owner'] },\n },\n // No `actions` buttons: a per-record button writes with the caller's\n // key, and the browser key here holds only functions:invoke by design.\n // Every screen and button in the console calls a function instead.\n },\n\n change_requests: {\n singular: 'change_request',\n fields: {\n kind: { type: 'string', required: true, indexSlot: 's1',\n validation: { enum: ['credit_grant', 'plan_change', 'suspend', 'reactivate', 'data_fix'] } },\n state: { type: 'string', required: true, indexSlot: 's2',\n validation: { enum: ['pending', 'approved', 'rejected', 'expired', 'applied', 'failed'] } },\n requested_by: { type: 'string', required: true, indexSlot: 's3' }, // end-user id, set by the function\n decided_by: { type: 'string', indexSlot: 's4' }, // end-user id, or 'policy'\n amount: { type: 'int', indexSlot: 'n1' }, // credits for credit_grant, else 0\n // `restrict`: an account with change requests cannot be deleted out\n // from under its own history (close it instead).\n account_ref: { type: 'relation', relationTo: 'accounts', onDelete: 'restrict', indexed: true },\n requested_at: { type: 'datetime', indexSlot: 't1' },\n decided_at: { type: 'datetime', indexSlot: 't2' },\n // `<Idempotency-Key>:<account id>` \u2014 unique, so a retried request\n // finds the first row instead of filing (or applying) twice\n request_key: { type: 'string', unique: true },\n payload: { type: 'json' }, // the exact patch to apply (allow-listed per kind)\n reason: { type: 'text', required: true },\n decision_note: { type: 'text' },\n error: { type: 'text' },\n auto: { type: 'bool' },\n },\n },\n\n // Thresholds are ROWS, not code: change \"auto-approve credit grants\n // under 500\" in the dashboard and the next request follows it.\n approval_policies: {\n singular: 'approval_policy',\n fields: {\n kind: { type: 'string', required: true, unique: true, indexSlot: 's1' },\n approver_permission: { type: 'string', indexSlot: 's2' }, // default changes.approve\n auto_approve_below: { type: 'int', indexSlot: 'n1', validation: { min: 0 } }, // absent = never auto\n // the CUMULATIVE caps: a per-grant threshold alone is bypassed by\n // splitting one big grant into many small ones. Rolling 24 h:\n // daily_cap \u2192 SUM of auto-applied amounts, per requester AND per\n // account (absent = the threshold itself);\n // daily_count \u2192 NUMBER of auto-applied grants per requester, held\n // under a lock, so racing calls cannot overshoot the\n // sum cap by more than this many grants (absent = 25).\n // Past either cap, the grant is filed pending for an approver.\n auto_approve_daily_cap: { type: 'int', validation: { min: 0 } },\n auto_approve_daily_count: { type: 'int', validation: { min: 1 } },\n expires_after_hours: { type: 'int', indexSlot: 'n2', validation: { min: 0 } }, // 0 = never expires\n enabled: { type: 'bool' },\n },\n },\n },\n },\n\n functions: {\n // File one change request per target account (1\u201325 = a bulk fix). Small\n // credit grants under the policy threshold apply at once, atomically with\n // their request row; everything else waits for an approver.\n 'request-change': {\n entry: './functions/request-change.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write', 'orgs:read', 'notifications:send'],\n egressAllow: [],\n signature: {\n input: {\n kind: \"'credit_grant' | 'plan_change' | 'suspend' | 'reactivate' | 'data_fix'\",\n account_ids: 'string[]',\n amount: 'number | undefined',\n payload: 'Record<string, unknown> | undefined',\n reason: 'string',\n },\n output: '{ results: Array<{ account_id: string; request_id?: string; state: string; error?: string }> }',\n },\n },\n\n // Approve or reject ONE pending request. Refuses the requester (403\n // self_approval; the hook is the backstop). An approval applies the change\n // in the same transaction, guarded on `state == pending`.\n 'decide-change': {\n entry: './functions/decide-change.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write', 'orgs:read', 'notifications:send'],\n egressAllow: [],\n signature: {\n input: { request_id: 'string', decision: \"'approve' | 'reject'\", note: 'string | undefined' },\n output: '{ request_id: string; state: string; error?: string }',\n },\n },\n\n // EVERY staff read (account search, one account, the request queue, the\n // policies, an account's notes) plus writing a note, each op behind the\n // orgs permission check. The browser key holds only functions:invoke, so\n // a signed-in person outside the staff org reads nothing (see the README).\n 'console-api': {\n entry: './functions/console-api.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'orgs:read', 'comments:read', 'comments:write'],\n egressAllow: [],\n signature: {\n input: {\n op: \"'search' | 'account' | 'requests' | 'policies' | 'notes' | 'add_note' | undefined\",\n q: 'string | undefined',\n by: \"'email' | 'external_ref' | 'name' | undefined\",\n status: 'string | undefined',\n plan: 'string | undefined',\n sort: 'string | undefined',\n id: 'string | undefined',\n state: 'string | undefined',\n account_id: 'string | undefined',\n mine: 'boolean | undefined',\n body: 'string | undefined',\n cursor: 'string | undefined',\n limit: 'number | undefined',\n },\n output: 'unknown',\n },\n },\n\n // The change history of one account (or one request): the human-attributed\n // change requests, plus the platform audit rows that name the record.\n 'audit-trail': {\n entry: './functions/audit-trail.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'orgs:read'],\n // An API key of THIS backend holding `features:read`, because the audit\n // stream is not on the function's scoped callback. Be clear about what\n // that key is: `features:read` is NOT audit-only. It also opens every\n // feature's data export (all tenant rows, end users, transcripts,\n // function source), so it is a FULL-PROJECT EXPORT CREDENTIAL. It lives\n // only in this function's secret: never in a browser bundle, a client\n // app, a log line or a second function.\n secrets: ['secret:vxil_read_key'],\n egressAllow: [],\n signature: {\n input: { subject: 'string', cursor: 'string | undefined' },\n output: '{ requests: unknown[]; events: unknown[]; next_cursor: string | null; scanned: number; truncated: boolean }',\n },\n },\n\n // Hourly: pending requests older than their policy's expires_after_hours\n // become `expired`, and the requester is told.\n 'expire-pending': {\n entry: './functions/expire-pending.ts',\n trigger: { kind: 'cron', schedule: '0 * * * *' },\n scopes: ['cms:read', 'cms:write', 'notifications:send'],\n egressAllow: [],\n },\n },\n\n // References only \u2014 values are stored once and never appear in this file.\n secrets: {\n oidc_client_secret: { feature: 'auth', description: 'OIDC client secret for your staff identity provider' },\n vxil_read_key: { feature: 'functions', description: 'a vxil API key holding features:read (a full-project export credential, not audit-only) \u2014 the audit-trail function reads the audit stream with it; never put it in a browser' },\n },\n\n seed: {\n cms: [\n {\n collection: 'approval_policies',\n items: [\n // credit grants under 500 units apply at once, up to 5,000 units (and 25\n // grants) a day per requester and 5,000 a day per account; past that, or\n // at 500 and above, they wait 72 h for an approver\n { kind: 'credit_grant', approver_permission: 'changes.approve', auto_approve_below: 500, auto_approve_daily_cap: 5000, auto_approve_daily_count: 25, expires_after_hours: 72, enabled: true },\n // a plan override always needs an approver (no auto_approve_below)\n { kind: 'plan_change', approver_permission: 'changes.approve', expires_after_hours: 48, enabled: true },\n ],\n },\n {\n collection: 'accounts',\n items: [\n { name: 'Acme Analytics', email: 'ops@acme-analytics.example', plan: 'growth', status: 'active', credits: 12000, seats: 25, created_at: '2025-11-03T09:00:00Z', last_active_at: '2026-09-30T16:12:00Z', external_ref: 'cus_1001', notes_count: 0, billing_contact: 'finance@acme-analytics.example', risk_flags: [] },\n { name: 'Bluebird Labs', email: 'admin@bluebird.example', plan: 'starter', status: 'active', credits: 800, seats: 5, created_at: '2026-02-14T10:30:00Z', last_active_at: '2026-10-01T08:45:00Z', external_ref: 'cus_1002', notes_count: 0, billing_contact: 'ap@bluebird.example', risk_flags: [] },\n { name: 'Cobalt Freight', email: 'it@cobalt-freight.example', plan: 'enterprise', status: 'active', credits: 250000, seats: 140, created_at: '2024-06-20T12:00:00Z', last_active_at: '2026-10-02T11:20:00Z', external_ref: 'cus_1003', notes_count: 0, billing_contact: 'procurement@cobalt-freight.example', risk_flags: ['high_usage'] },\n { name: 'Driftwood Studio', email: 'hello@driftwood.example', plan: 'starter', status: 'suspended', credits: 0, seats: 2, created_at: '2026-05-01T15:00:00Z', last_active_at: '2026-08-11T19:05:00Z', external_ref: 'cus_1004', notes_count: 0, billing_contact: 'hello@driftwood.example', risk_flags: ['chargeback_history'] },\n { name: 'Evergreen Health', email: 'platform@evergreen.example', plan: 'growth', status: 'active', credits: 4300, seats: 32, created_at: '2025-08-09T07:30:00Z', last_active_at: '2026-09-29T13:40:00Z', external_ref: 'cus_1005', notes_count: 0, billing_contact: 'billing@evergreen.example', risk_flags: [] },\n ],\n },\n ],\n },\n});\n",
18286
+ "readme": "# Admin console (saas)\n\nThe back-office every internal-tools team builds first. Staff sign in through **your company identity\nprovider**, look up and edit customer accounts, and fix up to 25 accounts at once. Sensitive changes\n(usage-credit grants, plan overrides, suspensions, data fixes) become **requests that someone else\napproves**. Each account shows its change history and internal notes.\n\nApprovals here are **rows plus one deciding function**: a state on a record, a hook that guards its\ntransitions, and one function that decides. There is no approval engine and no visual flow.\n\n```bash\nmkdir my-console && cd my-console\nvxil init --template admin-console # init scaffolds into the CURRENT directory\nvxil quickstart --env staging --no-push # or `vxil link <slug> --env staging`\nprintf '%s' \"$OIDC_SECRET\" | vxil secrets set auth/oidc_client_secret\n# the audit-trail function's key. features:read is a FULL-PROJECT EXPORT credential, not\n# audit-only: it lives in this one function secret and never in a browser (see \"Audit trail\")\nvxil login # keys mint needs a dashboard session, not a project key\nvxil keys mint --name console-audit-read --scopes features:read --json | jq -r .api_key | vxil secrets set functions/vxil_read_key\nvxil push\n```\n\n`vxil link --key` stores the key you give it in `~/.vxil/credentials.json` (so does `vxil login` with its session): treat that file as a secret \u2014 never commit it, copy it into an image or share it.\n\nBefore you push, edit three values in `vxil.config.ts`: `issuer`, `clientId` and `allowedDomains` in\n`features.auth.providers.oidc`. Also set the console URL in `allowedRedirectOrigins` and in\n`CONSOLE_URL` at the top of `functions/request-change.ts`, `decide-change.ts` and\n`expire-pending.ts`, which is the link in every approval message.\n\n> **Plan note.** Five functions and one hourly cron fit the Free plan's counts (5 functions, 3 cron\n> triggers). The functions deploy on Free when the project's workload is `staging` or `development`\n> (`vxil projects workload <slug> development`). On a Free `production` project, `vxil push` stops\n> before it writes anything and names the ways out: change the workload, upgrade to Developer, or run\n> `vxil push --skip-functions`.\n\n## What it provisions\n\n| Collection | What a row is | Teaches |\n|---|---|---|\n| `accounts` | one customer account: name, e-mail (unique), plan, status, **credits** (usage units, never money), seats, your product's own id (`external_ref`) | list, filter and sort on index slots; a status state machine; two **role-gated fields** |\n| `change_requests` | one requested change to one account: kind, state, who asked, who decided, the exact patch, the reason | approvals as data; self-approval impossible by construction; compare-and-set decisions |\n| `approval_policies` | one row per kind: who may approve it, the auto-approve threshold, when a request expires | thresholds as rows, not code |\n\n| Function | Trigger | What it does |\n|---|---|---|\n| `request-change` | http | files one request per target account (1\u201325), or applies it at once under the policy threshold |\n| `decide-change` | http | approves or rejects one pending request; an approval applies it in the same transaction |\n| `console-api` | http | every staff read (account search and lookup, the request queue, the policies, an account's notes) and writing a note, each behind the permission check |\n| `audit-trail` | http | one account's (or one request's) history: the requests and the platform audit rows |\n| `expire-pending` | cron, hourly | pending requests past their policy's deadline become `expired` |\n\nPlus internal notes (comments, through `console-api`) and approval messages (inbox and e-mail).\n\n## Set up the staff organization (once, over the API)\n\nRoles are rows, so this needs no deploy. Use a server key (`vxil api` uses the CLI's key):\n\n```bash\nvxil api POST /v1/orgs/roles --data '{\"role_key\":\"support\",\"name\":\"Support\",\"permissions\":[\"accounts.read\",\"changes.request\",\"notes.write\"]}'\nvxil api POST /v1/orgs/roles --data '{\"role_key\":\"approver\",\"name\":\"Approver\",\"permissions\":[\"accounts.read\",\"changes.approve\"]}'\nvxil api POST /v1/orgs/roles --data '{\"role_key\":\"console-admin\",\"name\":\"Console admin\",\"permissions\":[\"accounts.read\",\"changes.request\",\"changes.approve\",\"notes.write\",\"audit.read\",\"policies.manage\"]}'\n```\n\nEach person signs in once through the identity provider, which creates their user. Then create the\nstaff organization with its first owner and seat everyone else on a role:\n\n```bash\nvxil api GET \"/v1/users?email=lead@example.com\" # \u2192 the user id\nvxil api POST /v1/orgs --data '{\"slug\":\"staff\",\"name\":\"Staff\",\"owner_user_id\":\"<lead>\"}'\nvxil api POST /v1/orgs/org_\u2026/members --data '{\"user_id\":\"<sam>\",\"role\":\"support\"}'\nvxil api POST /v1/orgs/org_\u2026/members --data '{\"user_id\":\"<ana>\",\"role\":\"approver\"}'\nvxil api POST /v1/orgs/org_\u2026/members --data '{\"user_id\":\"<cat>\",\"role\":\"console-admin\"}'\n```\n\nThe org `owner` holds every permission (`*`). Keep that role for one or two break-glass people.\n\nRun the console in **its own project**, holding only staff. Every function looks up the caller's\nactive organization and asks the orgs permission check inside it, so a project that also holds\nyour customers' organizations would let a customer org's owner pass those checks in their own org.\n\n## Who can sign in, and why the browser reads nothing directly\n\n**`allowedDomains` lets in everyone at your company domain, not only staff.** Any colleague\nwith an `@example.com` identity can complete sign-in and get a session. Two things keep that\nfrom becoming access to customer data:\n\n1. **Assign the console's app in your identity provider to the staff group only** (in most\n providers: the app's assignment or access policy). Then a colleague outside the group is\n stopped at the provider and never gets a session. Do this first.\n2. **The browser key reads nothing on its own.** Even with a session, every read and write goes\n through a function that asks the orgs permission check first, so a signed-in person who is not\n in the staff org gets `403` from every function and sees no account, request or note.\n\nThe console's browser bundle holds two keys:\n\n| Key | Class | Scopes |\n|---|---|---|\n| `console-auth` | server | `auth:signin` (starting the identity-provider sign-in, refreshing the session) |\n| `console-data` | **public / thin-client** (dashboard \u2192 API keys) | `functions:invoke` **only** |\n\n`console-data` is minted as the thin-client class, so the edge refuses any request on it that does\nnot carry a signed-in session. A copy lifted out of the bundle calls nothing.\n\n**Why the browser key holds no `cms:read` or `comments:*`.** None of these collections has an\nowner field: every staff member works on every account. A collection read is checked against the\nkey's scopes, not against the staff org, so a browser key holding `cms:read` would let *anyone who\ncan sign in* (the whole domain, if step 1 is skipped) list every customer account with its e-mail.\nA key holding `comments:read` / `comments:write` would let them read and write internal notes. A key\nholding `cms:write` would let any staff member write any row with no permission check, no approval\nand no request row. So every screen calls `console-api` (reads and notes) or one of the change\nfunctions, and each one asks the orgs permission check (revocation-grade, so a removed role takes\neffect on the next call, not at the next sign-in). For the same reason the collections declare no\nper-record action buttons: a button writes with the caller's key.\n\n**Keep IdP group names distinct from the console's role names.** The `groups` claim rides the\nsession as roles next to the org role. A group literally named `console-admin` or `approver` would\npass the `readRoles` field gates below. The permission checks in the functions read the org\nmembership, not the group, so they are not affected.\n\n## Back-office over your own data\n\n- **List, filter, sort.** Every column a staff screen filters or sorts on is in an index slot:\n `status`, `plan`, `credits`, `seats`, `created_at`, `last_active_at`. For example, \"suspended\n accounts on growth, most credits first\" is\n `console-api` `{ op: 'search', status: 'suspended', plan: 'growth', sort: '-credits' }`, which the\n function runs as one indexed `GET /v1/cms/items/accounts?filter=\u2026&sort=-credits`. Sorted lists page\n with `cursor`, the same as the default order.\n- **Lookup.** `console-api` `{ op: 'search', q, by }`: `email` is an exact match on the unique slot,\n `external_ref` an exact match on an indexed field (your product's own id, without spending a\n slot), and `name` a case-sensitive `$startsWith` prefix. `{ op: 'account', id }` reads one account;\n `{ op: 'requests', state: 'pending' }` is the approver's queue (`account_id` narrows it to one\n account, `mine: true` to your own requests).\n- **Bulk fixes.** `request-change` takes up to 25 `account_ids`, which is one batch call of up to 25\n operations. Each account gets its own request row and its own result: `pending`, `applied`,\n `duplicate`, or `rejected` with `account_not_found`.\n- **Atomic multi-record changes.** An approval writes the request and the account in one atomic\n batch (`POST /v1/cms/batch` with `atomic: true`). Either both change or neither does.\n\n## Approvals as data, plus one function\n\n**The state machine is a hook.** `request_stage` allows only `pending \u2192 approved | rejected |\nexpired` and `approved \u2192 applied | failed`. Every other state is terminal. `request_born` allows a\nrequest to start only as `pending`, or as `applied` when a policy auto-approved it.\n\n**Self-approval is impossible by construction.** Write hooks see the row, not the caller, so the\nhuman ids are fields. `request-change` sets `requested_by` from the verified session, and\n`decide-change` sets `decided_by` the same way. The `no_self_approval` hook refuses any move into\n`approved` or `applied` where the two are equal. `decide-change` refuses the requester first with\n`403 self_approval`. The hook is the backstop, so a bug in the function cannot open the hole.\n\n**Auto-approve thresholds are rows, not code.** The seed has two policies:\n\n| kind | approver_permission | auto_approve_below | expires_after_hours |\n|---|---|---|---|\n| `credit_grant` | `changes.approve` | 500 | 72 |\n| `plan_change` | `changes.approve` | (none: never auto) | 48 |\n\nA credit grant of 499 or fewer units applies at once, as an `applied` request with `auto: true` and\n`decided_by: 'policy'`, written in the same atomic batch as the account change. A grant of 500 or\nmore waits for an approver. Edit the row and the next request follows it. A kind with no row never\nauto-applies and expires after 72 hours. Only `credit_grant` can auto-apply, because only it\ncarries an amount.\n\n**The threshold is also a daily budget, so splitting a grant does not skip the approver.** A\nper-grant threshold on its own is easy to get around: twenty grants of 499 to the same account are\n9,980 credits that nobody approved. So the `credit_grant` row also carries two rolling 24-hour caps:\n\n| field | seed | what it caps |\n|---|---|---|\n| `auto_approve_daily_cap` | 5,000 | the **sum** of auto-applied grants per requester, and separately per account |\n| `auto_approve_daily_count` | 25 | the **number** of auto-applied grants per requester |\n\nBefore it auto-applies anything, `request-change` adds up what was auto-applied in the last 24 hours\n(one aggregate over the requester's `decided_by: 'policy'` rows, one bounded read for the target\naccounts) and keeps a running total across a bulk call. A grant that would cross either sum cap is\nfiled `pending` with `reason: 'daily_auto_cap'` in its result, and the approvers are told as usual.\nIf a total cannot be read, the grant is filed `pending` too (fail closed). Leave\n`auto_approve_daily_cap` out and it defaults to the threshold itself (one threshold's worth a day).\n\nThe totals are a read, so two calls sent at the same instant could both pass them. The count cap is\nthe backstop for that: each auto-applied row is written under one lock per requester with a guard\nthat allows at most `auto_approve_daily_count` such rows in 24 hours. Racing calls can overshoot the\nsum cap by at most that many sub-threshold grants (25 \xD7 499 with the seed), never without bound.\nLower the count if that worst case is too high for you.\n\n**Two approvers can never both apply.** The decision's first write is conditional\n(`if: { state: 'pending' }`). Two approvers pressing at once both read `pending`. The second one's\nbatch finds the state already `approved` and gets a clean `409 already_decided`. The account\nchanges once.\n\n**A refused change is recorded.** When the account itself refuses the write (a closed account, a\ngrant that would take credits below zero, an e-mail another account already uses), the batch rolls\nback. A second batch then records the decision and the refusal: `pending \u2192 approved \u2192 failed`, with\nthe error on the row.\n\n**Expiry.** `expire-pending` runs hourly. Pending requests older than their kind's\n`expires_after_hours` become `expired` (0 means never), and the requester is told. Each write is\nconditional on `pending`, so a request decided in the same instant keeps its decision.\n\n**The payload is allow-listed per kind.** `request-change` refuses (`422 invalid_payload`) any\n`payload` key outside this list, before anything is written. `decide-change` applies the stored\npayload through the same list again.\n\n| kind | payload | applied as |\n|---|---|---|\n| `credit_grant` | none (`amount` 1\u20131,000,000) | `$inc: { credits: amount }` |\n| `plan_change` | `{ plan }` | `{ plan }` |\n| `suspend` / `reactivate` | none | `{ status: 'suspended' }` / `{ status: 'active' }` |\n| `data_fix` | any of `{ name, email, seats }` | those fields |\n\n**Retries.** Send an `Idempotency-Key` header with each button press. The function writes\n`<Idempotency-Key>:<account id>` into the request's unique `request_key` field, so a retried press\nfinds the first row (`state: 'duplicate'`) and never files or applies twice. The cms routes do\n**not** honor the `Idempotency-Key` header themselves. That unique field is the dedupe. The\nnotification sends **do** honor it, so each approver, and each requester per decision, is messaged\nonce. Without the header, every call counts as a new request.\n\n**Multi-step approvals are more data, not a graph.** For \"two approvals above 10,000 credits\", add a\nsecond state (`approved_1`) and a second permission (`changes.approve_large`) to the hook and the\npolicy row. `decide-change` stays one function that moves one state.\n\n## Who sees which field\n\n`risk_flags` (json) declares `readRoles: ['console-admin', 'owner']`, and `billing_contact` declares\n`['console-admin', 'approver', 'owner']`. `console-api` reads with the caller's session, so a support\nmember receives accounts **without** those fields: they are absent, not null, and cannot be used in a\nfilter or a sort. The role comes from the session (`orgClaims`, plus the IdP `groups`; see the\ngroup-name note above), so a role change applies at the member's next sign-in or refresh. Your own\nserver key still sees every field.\n\n## Internal notes\n\nNotes are comments on the topic `accounts:<item_id>`, read and written through `console-api`:\n\n```ts\nconst fn = vx.asEndUser(sessionToken).fn['console-api'];\nawait fn({ op: 'add_note', account_id: id, body: 'Customer asked to pause billing until May.' }); // needs notes.write\nconst { notes } = await fn({ op: 'notes', account_id: id }); // needs accounts.read\n```\n\nThe function writes the note with the caller's session, so the author is always the signed-in staff\nmember (a different `author_id` is refused). `notes_count` is a plain field; bump it from a function\nholding `cms:write` if a screen needs the count without loading the thread.\n\n## Audit trail per record\n\n`audit-trail` takes `{ subject: 'accounts:<id>' }` (or `'change_requests:<id>'`) and needs\n`audit.read`. It returns two halves:\n\n- **requests** are the change requests for the account: who asked, who decided, why, the exact\n patch, and the outcome. This half names the people.\n- **events** are the platform's own audit rows for the record: every create, update (with the\n incremented fields of a credit grant) and delete. They are kept 90 days on Free and 365 days on\n paid plans. Every one of them is recorded as **the project** (actor `tenant`), not as the staff\n member who pressed the button: the audit row names the record, never the person. The person is\n recorded in the request row (`requested_by`, `decided_by`), which is why the request rows exist.\n A note is audited the same way (`comment.created`, as the project); its author is on the note.\n\nThe audit stream is not on the function's scoped callback, so the function reads it with\n`vxil_read_key`, an API key holding `features:read`. It scans the 1,000 most recent tenant\nevents per call. `next_cursor` continues the scan, and `truncated: true` says there is more.\nReads are not audited.\n\n> **`vxil_read_key` is a full-project export credential, not an audit-only key.** `features:read`\n> also opens every feature's data export: all tenant rows, the end-user records, assistant\n> transcripts and function source. Treat it like the database password. It is stored only as this\n> one function's secret and is never sent to the browser, so a staff session cannot see it. Never\n> put it in a client bundle, a log line, or a second function, and rotate it if it ever leaves the\n> secret store. If you do not need the platform half of the trail, delete the `events` half and the\n> secret, and keep the request rows.\n\n## If your customer records live in your own database\n\nKeep the requests, approvals, policies, notes and messages here, and move two things:\n\n1. **`console-api`'s `search` and `account` ops**: keep their request and response shape, and replace the bodies with one HTTPS\n call to your database's query endpoint. The pattern is a function, a secret and one egress host\n (`examples/byo-db`).\n2. **The apply step in `decide-change`**: replace the batch's account write with a call to your own\n API. Then `approved` becomes a resting state (the decision is recorded), and your API's answer\n moves the request to `applied` or `failed`.\n\n## What this is not\n\n- **Not a generic approval engine.** There is no rule language and no designer. The rules are one\n hook, one policy row per kind, and a function you read top to bottom.\n- **Not a visual flow.** A second step is a second state and a second permission, in data.\n- **Not a payments surface.** Credits are usage units your product consumes. To refund money\n through your own payment provider, add a function with `payments:write` that calls\n `POST /v1/payments/refunds` (the payments integration: your provider account, never in the flow of\n funds), and file it as one more request kind.\n\n## Pairs with\n\n- `templates/team-workspace`: the same orgs, roles and `readRoles` machinery for your customers'\n teams.\n- `templates/helpdesk`: the customer-facing side that these staff screens answer.\n- `examples/apps/helpdesk` and `examples/apps/microblog`: a browser app with the same two-key split\n and sign-in shell to build the console UI on.\n",
18287
+ "functions": {
18288
+ "audit-trail.ts": "// audit-trail.ts \u2014 THE HISTORY OF ONE RECORD (a vxil function, http, end-user mode).\n//\n// POST /v1/fn/audit-trail { subject: 'accounts:<id>' | 'change_requests:<id>', cursor? }\n//\n// Two halves, because they answer different questions:\n//\n// requests \u2014 WHO asked for what, WHO decided, WHY. The change_requests rows\n// for the account (or the one request): requester, approver,\n// reason, decision note, the exact patch, the outcome. Every\n// change made through this console has one, so this is the\n// human-attributed trail.\n// events \u2014 WHAT the platform recorded on the record itself: every create,\n// update (with the incremented fields for a credit grant) and\n// delete, from the tenant audit stream, kept 90 days on Free and\n// 365 days on paid plans. These rows name the record but not the\n// human: every write is audited as the project (actor `tenant`),\n// not as the staff member, which is exactly why the request rows\n// above (requested_by / decided_by) exist.\n//\n// The audit stream is not reachable through the function's scoped callback, so\n// it is read with `vxil_read_key`: an API key of THIS backend holding\n// `features:read`. That scope is NOT audit-only: it also opens every feature's\n// data export, so the key is a FULL-PROJECT EXPORT CREDENTIAL. It lives only in\n// this function's secret; it is never returned, logged, or sent to a browser.\n// The stream is filtered here (newest first, a bounded number\n// of pages per call; `next_cursor` continues the scan, `truncated` says there\n// is more). Reads are not audited.\n//\n// Requires `audit.read` (console-admin).\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\nconst AUDIT_PAGE = 200; // the audit list's own maximum page\nconst MAX_AUDIT_PAGES = 5; // \u22641,000 most recent tenant events scanned per call\nconst MAX_REQUESTS = 50;\nconst COLLECTIONS = new Set(['accounts', 'change_requests']);\n\ninterface Payload { subject?: unknown; cursor?: unknown }\ntype Env = HttpFunctionEnvelope<Payload>;\ninterface AuditRow { id: number | string; event: string; actor: string; created_at: string; payload?: Record<string, unknown> | null }\n\nconst hdrs = (jwt: string) => ({ authorization: `Bearer ${jwt}`, 'content-type': 'application/json' });\nconst enc = (o: unknown) => encodeURIComponent(JSON.stringify(o));\nconst fail = (status: number, error: string, extra: Record<string, unknown> = {}) =>\n Response.json({ error, ...extra }, { status });\n\nasync function requirePermission(base: string, orgs: string, userId: string, permission: string): Promise<string | Response> {\n const c = await fetch(`${base}/v1/orgs/session-claims?user_id=${encodeURIComponent(userId)}`, { headers: hdrs(orgs) });\n if (!c.ok) return fail(502, 'orgs_unavailable', { status: c.status });\n const orgId = ((await c.json()) as { data?: { org_id?: string | null } }).data?.org_id;\n if (!orgId) return fail(403, 'forbidden', { permission, message: 'you are not a member of the staff organization' });\n const r = await fetch(\n `${base}/v1/orgs/${encodeURIComponent(orgId)}/check?user_id=${encodeURIComponent(userId)}&permission=${encodeURIComponent(permission)}`,\n { headers: hdrs(orgs) },\n );\n if (!r.ok) return fail(502, 'orgs_unavailable', { status: r.status });\n const allowed = ((await r.json()) as { data?: { allowed?: boolean } }).data?.allowed === true;\n return allowed ? orgId : fail(403, 'forbidden', { permission, message: `you need the '${permission}' permission` });\n}\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const { cms, orgs } = env.scoped_jwts ?? {};\n const readKey = env.secrets?.vxil_read_key;\n const me = env.end_user?.id;\n if (!me) return fail(401, 'sign_in_required', { message: 'call this with a signed-in staff session' });\n if (!cms || !orgs) return fail(500, 'missing_scopes');\n if (!readKey) return fail(409, 'missing_secret', { message: 'store an API key holding features:read as functions/vxil_read_key (a full-project export credential: keep it only in this secret)' });\n\n const m = typeof env.payload?.subject === 'string' ? /^([a-z_]+):([A-Za-z0-9_-]{1,64})$/.exec(env.payload.subject) : null;\n if (!m || !COLLECTIONS.has(m[1]!)) {\n return fail(422, 'invalid_payload', { message: \"subject must be 'accounts:<id>' or 'change_requests:<id>'\" });\n }\n const [, collection, itemId] = m as unknown as [string, string, string];\n\n const orgId = await requirePermission(base, orgs, me, 'audit.read');\n if (orgId instanceof Response) return orgId;\n\n // \u2500\u2500 half 1: the human-attributed change requests \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n let requests: unknown[] = [];\n if (collection === 'accounts') {\n const r = await fetch(\n `${base}/v1/cms/items/change_requests?filter=${enc({ account_ref: itemId })}&limit=${MAX_REQUESTS}`,\n { headers: hdrs(cms) },\n );\n if (r.ok) {\n const items = ((await r.json()) as { data?: { items?: Array<{ item_id: string; data: Record<string, unknown> }> } }).data?.items ?? [];\n requests = items.map((i) => ({ request_id: i.item_id, ...i.data }));\n }\n } else {\n const r = await fetch(`${base}/v1/cms/items/change_requests/${itemId}`, { headers: hdrs(cms) });\n if (r.ok) {\n const i = ((await r.json()) as { data?: { item_id: string; data: Record<string, unknown> } }).data;\n if (i) requests = [{ request_id: i.item_id, ...i.data }];\n }\n }\n\n // \u2500\u2500 half 2: the platform audit rows that name this record \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n const events: Array<{ id: string; event: string; actor: string; at: string; fields: string[] | null }> = [];\n let cursor = typeof env.payload?.cursor === 'string' && /^\\d+$/.test(env.payload.cursor) ? env.payload.cursor : null;\n let scanned = 0;\n for (let page = 0; page < MAX_AUDIT_PAGES; page++) {\n const r = await fetch(`${base}/v1/audit?limit=${AUDIT_PAGE}${cursor ? `&cursor=${cursor}` : ''}`, {\n headers: { authorization: `Bearer ${readKey}` },\n });\n if (!r.ok) return fail(502, 'audit_unavailable', { status: r.status });\n const body = ((await r.json()) as { data?: { events?: AuditRow[]; next_cursor?: string | null } }).data;\n const rows = body?.events ?? [];\n scanned += rows.length;\n for (const row of rows) {\n const pl = row.payload ?? {};\n if (pl.collection !== collection || pl.item_id !== itemId) continue;\n const inc = pl.inc && typeof pl.inc === 'object' ? Object.keys(pl.inc as object) : null;\n events.push({ id: String(row.id), event: row.event, actor: row.actor, at: row.created_at, fields: inc });\n }\n cursor = body?.next_cursor ?? null;\n if (!cursor) break;\n }\n\n return Response.json({ subject: `${collection}:${itemId}`, requests, events, next_cursor: cursor, scanned, truncated: cursor !== null });\n },\n};\n",
18289
+ "console-api.ts": "// console-api.ts \u2014 EVERY STAFF READ, PLUS INTERNAL NOTES (a vxil function, http, end-user mode).\n//\n// POST /v1/fn/console-api { op, ... }\n//\n// op needs does\n// search accounts.read { q?, by?: 'email'|'external_ref'|'name', status?, plan?, sort?, cursor?, limit? }\n// account accounts.read { id } \u2192 one account\n// requests accounts.read { state?, account_id?, mine?, cursor?, limit? } \u2192 change requests, newest first\n// policies accounts.read {} \u2192 the approval policy rows\n// notes accounts.read { account_id, cursor? } \u2192 the account's internal notes\n// add_note notes.write { account_id, body } \u2192 one note, authored as the caller\n//\n// WHY ONE FUNCTION AND NOT A BROWSER KEY WITH cms:read. Sign-in is open to\n// everyone your identity provider lets through `allowedDomains`: the whole\n// company domain, not only the staff org. A browser key holding `cms:read` or\n// `comments:read` would let ANY of those people list every customer account\n// (e-mails included) and read or write internal notes, because collection\n// reads are not checked against the staff org. So the console's browser key\n// holds only `functions:invoke`, and every read comes through here, where the\n// orgs permission check runs first (revocation-grade: a removed role is\n// refused on the next call). A signed-in person outside the staff org gets\n// 403 on every op.\n//\n// The reads run with the CALLER's session, so field-level gates hold here too:\n// `risk_flags` reaches console-admins only, `billing_contact` approvers and\n// console-admins only.\n//\n// Lookups are index-served: e-mail = the unique `email` slot (exact); your\n// product's id = the indexed `external_ref` (exact); name = a `$startsWith`\n// prefix on the `name` slot (case-sensitive). With no `by`, a `q` holding an\n// '@' is an e-mail, otherwise a name prefix.\n//\n// This is also the seam for the day accounts move to YOUR database: keep the\n// request/response shape of `search` and `account` and swap their bodies for\n// one HTTPS call to your SQL endpoint (see examples/byo-db \u2014 a function + a\n// secret + one egress host). The console does not change.\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\nconst MAX_LIMIT = 100;\nconst MAX_NOTE = 4000;\nconst READ_OPS = new Set(['search', 'account', 'requests', 'policies', 'notes']);\n/** Sortable account columns: each one is on an index slot. A leading '-' sorts descending. */\nconst SORTS = new Set(['name', 'email', 'plan', 'status', 'credits', 'seats', 'created_at', 'last_active_at']);\nconst STATES = new Set(['pending', 'approved', 'rejected', 'expired', 'applied', 'failed']);\n\ninterface Payload {\n op?: unknown; q?: unknown; by?: unknown; status?: unknown; plan?: unknown; sort?: unknown; cursor?: unknown; limit?: unknown;\n id?: unknown; state?: unknown; account_id?: unknown; mine?: unknown; body?: unknown;\n}\ntype Env = HttpFunctionEnvelope<Payload>;\ntype Item = { item_id: string; data: Record<string, unknown> };\n\nconst hdrs = (jwt: string) => ({ authorization: `Bearer ${jwt}`, 'content-type': 'application/json' });\nconst enc = (o: unknown) => encodeURIComponent(JSON.stringify(o));\nconst fail = (status: number, error: string, extra: Record<string, unknown> = {}) =>\n Response.json({ error, ...extra }, { status });\nconst str = (v: unknown, max = 200): string | null =>\n typeof v === 'string' && v.trim() && v.length <= max ? v.trim() : null;\nconst itemId = (v: unknown): string | null =>\n typeof v === 'string' && /^[A-Za-z0-9_-]{1,64}$/.test(v) ? v : null;\n\nasync function requirePermission(base: string, orgs: string, userId: string, permission: string): Promise<string | Response> {\n const c = await fetch(`${base}/v1/orgs/session-claims?user_id=${encodeURIComponent(userId)}`, { headers: hdrs(orgs) });\n if (!c.ok) return fail(502, 'orgs_unavailable', { status: c.status });\n const orgId = ((await c.json()) as { data?: { org_id?: string | null } }).data?.org_id;\n if (!orgId) return fail(403, 'forbidden', { permission, message: 'you are not a member of the staff organization' });\n const r = await fetch(\n `${base}/v1/orgs/${encodeURIComponent(orgId)}/check?user_id=${encodeURIComponent(userId)}&permission=${encodeURIComponent(permission)}`,\n { headers: hdrs(orgs) },\n );\n if (!r.ok) return fail(502, 'orgs_unavailable', { status: r.status });\n const allowed = ((await r.json()) as { data?: { allowed?: boolean } }).data?.allowed === true;\n return allowed ? orgId : fail(403, 'forbidden', { permission, message: `you need the '${permission}' permission` });\n}\n\n/** One cms list call, answered as { items, next_cursor }. */\nasync function list(base: string, cms: string, collection: string, params: string[]): Promise<Response> {\n const r = await fetch(`${base}/v1/cms/items/${collection}?${params.filter(Boolean).join('&')}`, { headers: hdrs(cms) });\n if (!r.ok) return fail(r.status === 422 ? 422 : 502, 'read_failed', { detail: (await r.text()).slice(0, 300) });\n const page = ((await r.json()) as { data?: { items?: Item[]; next_cursor?: string | null } }).data;\n return Response.json({\n items: (page?.items ?? []).map((i) => ({ item_id: i.item_id, data: i.data })),\n next_cursor: page?.next_cursor ?? null,\n });\n}\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const { cms, orgs, comments } = env.scoped_jwts ?? {};\n const me = env.end_user?.id;\n if (!me) return fail(401, 'sign_in_required', { message: 'call this with a signed-in staff session' });\n if (!cms || !orgs || !comments) return fail(500, 'missing_scopes');\n\n const p = env.payload ?? {};\n const op = typeof p.op === 'string' ? p.op : 'search';\n if (!READ_OPS.has(op) && op !== 'add_note') {\n return fail(422, 'invalid_payload', { message: \"op must be one of search, account, requests, policies, notes, add_note\" });\n }\n const limit = Math.min(Math.max(Number.isInteger(p.limit) ? (p.limit as number) : 25, 1), MAX_LIMIT);\n const cursor = typeof p.cursor === 'string' && p.cursor ? `cursor=${encodeURIComponent(p.cursor)}` : '';\n\n // \u2500\u2500 the permission check comes before ANY read \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n const orgId = await requirePermission(base, orgs, me, op === 'add_note' ? 'notes.write' : 'accounts.read');\n if (orgId instanceof Response) return orgId;\n\n switch (op) {\n case 'search': {\n const q = str(p.q);\n const by = p.by ?? (q?.includes('@') ? 'email' : 'name');\n if (by !== 'email' && by !== 'external_ref' && by !== 'name') {\n return fail(422, 'invalid_payload', { message: \"by must be 'email', 'external_ref' or 'name'\" });\n }\n const filter: Record<string, unknown> = {};\n if (q) {\n if (by === 'email') filter.email = q;\n else if (by === 'external_ref') filter.external_ref = q;\n else filter.name = { $startsWith: q };\n }\n const status = str(p.status, 32);\n const plan = str(p.plan, 64);\n if (status) filter.status = status;\n if (plan) filter.plan = plan;\n let sort = '';\n if (p.sort !== undefined) {\n if (typeof p.sort !== 'string' || !SORTS.has(p.sort.replace(/^-/, ''))) {\n return fail(422, 'invalid_payload', { message: `sort must be one of ${[...SORTS].join(', ')} (with an optional leading '-')` });\n }\n sort = `sort=${encodeURIComponent(p.sort)}`;\n }\n return list(base, cms, 'accounts', [Object.keys(filter).length ? `filter=${enc(filter)}` : '', sort, `limit=${limit}`, cursor]);\n }\n\n case 'account': {\n const id = itemId(p.id);\n if (!id) return fail(422, 'invalid_payload', { message: 'id must be an account id' });\n const r = await fetch(`${base}/v1/cms/items/accounts/${id}`, { headers: hdrs(cms) });\n if (r.status === 404) return fail(404, 'account_not_found');\n if (!r.ok) return fail(502, 'read_failed', { status: r.status });\n const i = ((await r.json()) as { data?: Item }).data;\n return Response.json({ item_id: i?.item_id ?? id, data: i?.data ?? {} });\n }\n\n case 'requests': {\n const filter: Record<string, unknown> = {};\n if (p.state !== undefined) {\n if (typeof p.state !== 'string' || !STATES.has(p.state)) return fail(422, 'invalid_payload', { message: 'state is not a request state' });\n filter.state = p.state;\n }\n if (p.account_id !== undefined) {\n const a = itemId(p.account_id);\n if (!a) return fail(422, 'invalid_payload', { message: 'account_id must be an account id' });\n filter.account_ref = a;\n }\n if (p.mine === true) filter.requested_by = me;\n return list(base, cms, 'change_requests', [\n Object.keys(filter).length ? `filter=${enc(filter)}` : '', 'sort=-requested_at', `limit=${limit}`, cursor,\n ]);\n }\n\n case 'policies':\n return list(base, cms, 'approval_policies', ['limit=100']);\n\n case 'notes':\n case 'add_note': {\n const accountId = itemId(p.account_id);\n if (!accountId) return fail(422, 'invalid_payload', { message: 'account_id must be an account id' });\n const topic = `accounts:${accountId}`;\n if (op === 'notes') {\n const r = await fetch(`${base}/v1/comments?topic=${encodeURIComponent(topic)}&limit=${limit}${cursor ? `&${cursor}` : ''}`, { headers: hdrs(comments) });\n if (!r.ok) return fail(502, 'read_failed', { status: r.status });\n const page = ((await r.json()) as { data?: { comments?: unknown[]; next_cursor?: string | null } }).data;\n return Response.json({ notes: page?.comments ?? [], next_cursor: page?.next_cursor ?? null });\n }\n const body = typeof p.body === 'string' ? p.body.trim() : '';\n if (!body || body.length > MAX_NOTE) return fail(422, 'invalid_payload', { message: `body is required (\u2264${MAX_NOTE} chars)` });\n // a note on an account that does not exist is refused, not orphaned\n const a = await fetch(`${base}/v1/cms/items/accounts/${accountId}`, { headers: hdrs(cms) });\n if (a.status === 404) return fail(404, 'account_not_found');\n if (!a.ok) return fail(502, 'read_failed', { status: a.status });\n // the author is the verified caller: a signed-in session cannot write as anyone else\n const w = await fetch(`${base}/v1/comments`, {\n method: 'POST',\n headers: hdrs(comments),\n body: JSON.stringify({ topic, body, author_id: me }),\n });\n if (!w.ok) return fail(w.status === 422 ? 422 : 502, 'note_failed', { status: w.status });\n const note = ((await w.json()) as { data?: unknown }).data;\n return Response.json({ note }, { status: 201 });\n }\n }\n return fail(422, 'invalid_payload');\n },\n};\n",
18290
+ "decide-change.ts": "// decide-change.ts \u2014 APPROVE OR REJECT ONE REQUEST (a vxil function, http, end-user mode).\n//\n// POST /v1/fn/decide-change { request_id, decision: 'approve' | 'reject', note? }\n//\n// 1. Reads the request; only a `pending` one can be decided (409 otherwise).\n// 2. Checks the caller holds the permission the kind's POLICY names\n// (`approver_permission`, default `changes.approve`) \u2014 403 naming it.\n// 3. Refuses the requester: 403 `self_approval`. The `no_self_approval` hook\n// refuses the same write again inside the database, so a bug here cannot\n// open the hole.\n// 4. Approve = ONE atomic batch, three writes:\n// change_requests/<id> if state == 'pending' \u2192 approved (decided_by = you)\n// accounts/<account> the stored, allow-listed patch ($inc credits, or fields)\n// change_requests/<id> if state == 'approved' \u2192 applied\n// All or nothing. Two approvers pressing at once: the second one's first\n// write finds the state no longer `pending` and gets a clean 409\n// `already_decided`; the account changes once.\n// If the ACCOUNT write is refused (a closed account, credits that would go\n// negative, a duplicate e-mail) the whole batch rolls back, and a second\n// small batch records the decision and the refusal: pending \u2192 approved \u2192\n// failed, with the error on the row.\n// 5. Reject = one conditional write, pending \u2192 rejected.\n// 6. Tells the requester (inbox + e-mail), keyed so a retry sends once.\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\nconst CONSOLE_URL = 'https://console.example.com';\nconst DEFAULT_EXPIRY_HOURS = 72;\n\ninterface Payload { request_id?: unknown; decision?: unknown; note?: unknown }\ntype Env = HttpFunctionEnvelope<Payload>;\ninterface ChangeRequest {\n kind: string; state: string; requested_by: string; amount?: number;\n account_ref: string; requested_at?: string; payload?: Record<string, unknown>; reason?: string;\n}\ninterface Policy { approver_permission?: string; expires_after_hours?: number | null }\ninterface BatchResult { op: string; status: number; data?: unknown; error?: { code?: string; message?: string } }\ntype Patch = { $inc: { credits: number } } | { data: Record<string, unknown> };\n\nconst hdrs = (jwt: string) => ({ authorization: `Bearer ${jwt}`, 'content-type': 'application/json' });\nconst enc = (o: unknown) => encodeURIComponent(JSON.stringify(o));\nconst fail = (status: number, error: string, extra: Record<string, unknown> = {}) =>\n Response.json({ error, ...extra }, { status });\n\n/** The SAME per-kind allow-list request-change enforces, re-applied to the\n * stored payload: what gets applied is only ever an allow-listed patch. */\nfunction patchFor(kind: string, amount: number, payload: Record<string, unknown>): Patch | string {\n const keys = Object.keys(payload);\n const extra = (allowed: string[]) => keys.filter((k) => !allowed.includes(k));\n switch (kind) {\n case 'credit_grant':\n return Number.isInteger(amount) && amount > 0 && !keys.length ? { $inc: { credits: amount } } : 'bad credit_grant';\n case 'plan_change':\n return typeof payload.plan === 'string' && !extra(['plan']).length ? { data: { plan: payload.plan } } : 'bad plan_change';\n case 'suspend':\n return keys.length ? 'bad suspend' : { data: { status: 'suspended' } };\n case 'reactivate':\n return keys.length ? 'bad reactivate' : { data: { status: 'active' } };\n case 'data_fix':\n return keys.length && !extra(['name', 'email', 'seats']).length ? { data: { ...payload } } : 'bad data_fix';\n default:\n return `unknown kind '${kind}'`;\n }\n}\n\nasync function requirePermission(base: string, orgs: string, userId: string, permission: string): Promise<string | Response> {\n const c = await fetch(`${base}/v1/orgs/session-claims?user_id=${encodeURIComponent(userId)}`, { headers: hdrs(orgs) });\n if (!c.ok) return fail(502, 'orgs_unavailable', { status: c.status });\n const orgId = ((await c.json()) as { data?: { org_id?: string | null } }).data?.org_id;\n if (!orgId) return fail(403, 'forbidden', { permission, message: 'you are not a member of the staff organization' });\n const r = await fetch(\n `${base}/v1/orgs/${encodeURIComponent(orgId)}/check?user_id=${encodeURIComponent(userId)}&permission=${encodeURIComponent(permission)}`,\n { headers: hdrs(orgs) },\n );\n if (!r.ok) return fail(502, 'orgs_unavailable', { status: r.status });\n const allowed = ((await r.json()) as { data?: { allowed?: boolean } }).data?.allowed === true;\n return allowed ? orgId : fail(403, 'forbidden', { permission, message: `you need the '${permission}' permission` });\n}\n\nasync function batch(base: string, cms: string, ops: unknown[]): Promise<{ ok: boolean; status: number; results: BatchResult[] }> {\n const r = await fetch(`${base}/v1/cms/batch`, {\n method: 'POST',\n headers: hdrs(cms),\n body: JSON.stringify({ atomic: true, ops }),\n });\n const out = ((await r.json().catch(() => ({}))) as { data?: { ok?: boolean; results?: BatchResult[] } }).data;\n return { ok: r.ok && out?.ok === true, status: r.status, results: out?.results ?? [] };\n}\n/** The first op that really failed (not the rolled-back / not-run ones). */\nconst firstFailure = (rs: BatchResult[]) =>\n rs.findIndex((x) => x.status >= 300 && x.error?.code !== 'rolled_back' && x.error?.code !== 'not_run');\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const { cms, orgs, notifications } = env.scoped_jwts ?? {};\n const me = env.end_user?.id;\n if (!me) return fail(401, 'sign_in_required', { message: 'call this with a signed-in staff session' });\n if (!cms || !orgs || !notifications) return fail(500, 'missing_scopes');\n\n const p = env.payload ?? {};\n const id = typeof p.request_id === 'string' && /^[A-Za-z0-9_-]{1,64}$/.test(p.request_id) ? p.request_id : null;\n const decision = p.decision === 'approve' || p.decision === 'reject' ? p.decision : null;\n const note = typeof p.note === 'string' ? p.note.slice(0, 2000) : null;\n if (!id || !decision) return fail(422, 'invalid_payload', { message: \"need { request_id, decision: 'approve' | 'reject', note? }\" });\n\n const got = await fetch(`${base}/v1/cms/items/change_requests/${id}`, { headers: hdrs(cms) });\n if (got.status === 404) return fail(404, 'not_found');\n if (!got.ok) return fail(502, 'cms_unavailable', { status: got.status });\n const cr = ((await got.json()) as { data?: { data?: ChangeRequest } }).data?.data;\n if (!cr) return fail(404, 'not_found');\n if (cr.state !== 'pending') return fail(409, 'already_decided', { state: cr.state });\n\n const pol = await fetch(`${base}/v1/cms/items/approval_policies?filter=${enc({ kind: cr.kind })}&limit=1`, { headers: hdrs(cms) });\n const policy: Policy = pol.ok\n ? (((await pol.json()) as { data?: { items?: Array<{ data?: Policy }> } }).data?.items?.[0]?.data ?? {})\n : {};\n const permission = policy.approver_permission || 'changes.approve';\n const orgId = await requirePermission(base, orgs, me, permission);\n if (orgId instanceof Response) return orgId;\n\n if (cr.requested_by === me) {\n return fail(403, 'self_approval', { message: 'you cannot decide a change you requested \u2014 ask another approver' });\n }\n const hours = policy.expires_after_hours ?? DEFAULT_EXPIRY_HOURS;\n if (hours > 0 && cr.requested_at && Date.parse(cr.requested_at) + hours * 3_600_000 < Date.now()) {\n return fail(409, 'expired', { message: `requests of this kind expire after ${hours} h; file a new one` });\n }\n\n const now = new Date().toISOString();\n const decided = { decided_by: me, decided_at: now, decision_note: note };\n let state: string;\n let error: string | undefined;\n\n if (decision === 'reject') {\n const r = await fetch(`${base}/v1/cms/items/change_requests/${id}`, {\n method: 'PATCH',\n headers: hdrs(cms),\n body: JSON.stringify({ if: { state: 'pending' }, data: { state: 'rejected', ...decided } }),\n });\n if (r.status === 409) return fail(409, 'already_decided');\n if (!r.ok) return fail(r.status === 422 ? 422 : 502, 'reject_refused', { detail: (await r.text()).slice(0, 300) });\n state = 'rejected';\n } else {\n const patch = patchFor(cr.kind, Number(cr.amount ?? 0), cr.payload ?? {});\n const approve = { op: 'patch', collection: 'change_requests', id, if: { state: 'pending' }, data: { state: 'approved', ...decided } };\n if (typeof patch === 'string') {\n // a stored payload outside the allow-list is never applied\n const rec = await batch(base, cms, [approve,\n { op: 'patch', collection: 'change_requests', id, if: { state: 'approved' }, data: { state: 'failed', error: patch } }]);\n if (!rec.ok) return fail(409, 'already_decided');\n state = 'failed';\n error = patch;\n } else {\n const res = await batch(base, cms, [\n approve,\n { op: 'patch', collection: 'accounts', id: cr.account_ref, ...patch },\n { op: 'patch', collection: 'change_requests', id, if: { state: 'approved' }, data: { state: 'applied' } },\n ]);\n if (res.ok) {\n state = 'applied';\n } else {\n const at = firstFailure(res.results);\n const why = res.results[at]?.error;\n if (at === 0) {\n // the guard on `pending` lost a race, or the hook refused the decision\n if (res.results[0]?.status === 409) return fail(409, 'already_decided');\n if (/self_approval/.test(why?.message ?? '')) return fail(403, 'self_approval');\n return fail(422, 'decision_refused', { detail: why?.message ?? why?.code });\n }\n if (at !== 1) return fail(502, 'apply_incomplete', { detail: why?.code ?? `http_${res.status}` });\n // the ACCOUNT refused the change: record the decision and the refusal\n error = `${why?.code ?? 'refused'}: ${(why?.message ?? '').slice(0, 300)}`;\n const rec = await batch(base, cms, [approve,\n { op: 'patch', collection: 'change_requests', id, if: { state: 'approved' }, data: { state: 'failed', error } }]);\n if (!rec.ok) return fail(409, 'already_decided');\n state = 'failed';\n }\n }\n }\n\n // \u2500\u2500 tell the requester, once per decision \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: { ...hdrs(notifications), 'idempotency-key': `decided:${id}` },\n body: JSON.stringify({\n user_id: cr.requested_by,\n template: 'transactional',\n channel: 'both',\n data: {\n subject: `Your ${cr.kind} request was ${state === 'applied' ? 'approved and applied' : state}`,\n paragraph: `Decided by ${me}${note ? ` \u2014 \"${note.slice(0, 300)}\"` : ''}.`\n + `${error ? ` It could not be applied: ${error}.` : ''} Details: ${CONSOLE_URL}/requests/${id}`,\n },\n }),\n }).catch(() => null);\n\n return Response.json({ request_id: id, state, ...(error ? { error } : {}) }, { status: state === 'failed' ? 422 : 200 });\n },\n};\n",
18291
+ "expire-pending.ts": "// expire-pending.ts \u2014 STALE REQUESTS EXPIRE (a vxil function, cron hourly).\n//\n// For every request kind: pending requests older than the kind's policy\n// `expires_after_hours` (72 h when the kind has no policy row; 0 = never)\n// become `expired`, and the requester is told once. A stale request never\n// lingers as something an approver could still apply weeks later.\n//\n// Each write is conditional (`if: { state: 'pending' }`): a request an approver\n// decides in the same instant keeps the approver's decision, and this tick\n// skips it. At most MAX_EXPIRE_PER_TICK rows per tick; `more: true` in the\n// result says the next tick continues (the rest are still pending and still\n// past their deadline, so they are still in the filter).\n\n// cron-walk: drains-filter \u2014 each expired request leaves the state=pending filter\n\nimport type { CronFunctionEnvelope } from '@vxil/sdk';\n\nconst MAX_EXPIRE_PER_TICK = 100;\nconst DEFAULT_EXPIRY_HOURS = 72;\nconst CONSOLE_URL = 'https://console.example.com';\nconst KINDS = ['credit_grant', 'plan_change', 'suspend', 'reactivate', 'data_fix'];\n\ninterface Policy { kind?: string; expires_after_hours?: number | null }\ninterface Row { item_id: string; data: { kind: string; requested_by: string; requested_at?: string } }\n\nconst hdrs = (jwt: string) => ({ authorization: `Bearer ${jwt}`, 'content-type': 'application/json' });\nconst enc = (o: unknown) => encodeURIComponent(JSON.stringify(o));\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as CronFunctionEnvelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const { cms, notifications } = env.scoped_jwts ?? {};\n if (!cms || !notifications) return Response.json({ error: 'missing_scopes' }, { status: 500 });\n\n const pr = await fetch(`${base}/v1/cms/items/approval_policies?limit=100`, { headers: hdrs(cms) });\n if (!pr.ok) return Response.json({ error: `policies ${pr.status}` }, { status: 502 });\n const policies = ((await pr.json()) as { data?: { items?: Array<{ data: Policy }> } }).data?.items ?? [];\n const hoursFor = (kind: string) => policies.find((x) => x.data.kind === kind)?.data.expires_after_hours ?? DEFAULT_EXPIRY_HOURS;\n\n const now = new Date();\n let expired = 0;\n let more = false;\n const failed: string[] = [];\n for (const kind of KINDS) {\n const hours = hoursFor(kind);\n if (!hours || hours <= 0) continue;\n const budget = MAX_EXPIRE_PER_TICK - expired;\n if (budget <= 0) { more = true; break; }\n const cutoff = new Date(now.getTime() - hours * 3_600_000).toISOString();\n const filter = { state: 'pending', kind, requested_at: { $lt: cutoff } };\n const r = await fetch(`${base}/v1/cms/items/change_requests?filter=${enc(filter)}&sort=requested_at&limit=${budget}`, {\n headers: hdrs(cms),\n });\n if (!r.ok) { failed.push(`${kind}: query ${r.status}`); continue; }\n const page = ((await r.json()) as { data?: { items?: Row[]; next_cursor?: string | null } }).data;\n const rows = page?.items ?? [];\n if (rows.length === budget && page?.next_cursor) more = true;\n\n for (const row of rows) {\n const w = await fetch(`${base}/v1/cms/items/change_requests/${row.item_id}`, {\n method: 'PATCH',\n headers: hdrs(cms),\n body: JSON.stringify({\n if: { state: 'pending' },\n data: { state: 'expired', decided_at: now.toISOString(), decision_note: `expired after ${hours} h without a decision` },\n }),\n });\n if (w.status === 409) continue; // decided in the meantime \u2014 the decision stands\n if (!w.ok) { failed.push(`${row.item_id}: ${w.status}`); continue; }\n expired++;\n await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: { ...hdrs(notifications), 'idempotency-key': `expired:${row.item_id}` },\n body: JSON.stringify({\n user_id: row.data.requested_by,\n template: 'transactional',\n channel: 'both',\n data: {\n subject: `Your ${row.data.kind} request expired`,\n paragraph: `Nobody decided it within ${hours} h, so it was closed without being applied. `\n + `File it again if it is still needed: ${CONSOLE_URL}/requests/${row.item_id}`,\n },\n }),\n }).catch(() => null);\n }\n }\n return Response.json({ expired, more, failed });\n },\n};\n",
18292
+ "request-change.ts": "// request-change.ts \u2014 FILE A CHANGE REQUEST (a vxil function, http, end-user mode).\n//\n// Called by the console with the signed-in staff member's session:\n// POST /v1/fn/request-change (Idempotency-Key: <one per button press>)\n// { kind, account_ids: [1\u201325 ids], amount?, payload?, reason }\n//\n// One request ROW per target account, so 25 ids = one bulk fix. For each:\n// \u2022 small credit grants under the policy threshold apply AT ONCE: the request\n// row (state 'applied', auto: true, decided_by 'policy') and the account\n// change commit in ONE atomic batch \u2014 both or neither \u2014 but only while the\n// policy's rolling 24 h caps hold: the SUM of auto-applied grants per\n// requester and per account (`auto_approve_daily_cap`) and the NUMBER per\n// requester (`auto_approve_daily_count`, enforced under a lock). A grant\n// that would cross a cap is filed 'pending' (reason 'daily_auto_cap'), so\n// splitting one big grant into many small ones still reaches an approver;\n// \u2022 everything else is filed 'pending' (one batch of \u226425 creates) and every\n// org member who may approve it gets one message (inbox + e-mail).\n//\n// What it refuses, before anything is written:\n// \u2022 a caller without `changes.request` in the staff org (403, naming it);\n// \u2022 a `payload` key outside the per-kind allow-list (422) \u2014 the stored patch\n// is exactly what an approver sees and exactly what gets applied.\n//\n// Retries: the row's unique `request_key` is `<Idempotency-Key>:<account id>`.\n// CMS does not honor the Idempotency-Key header itself; the unique field is the\n// dedupe, so the same button press retried files (or applies) nothing twice \u2014\n// the duplicate comes back as `state: 'duplicate'`. The notification sends DO\n// honor Idempotency-Key, keyed per approver. Without an Idempotency-Key header\n// the platform uses the request id, so every call is a fresh request.\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\n/** Where approvers open a request. Change to your console's URL. */\nconst CONSOLE_URL = 'https://console.example.com';\nconst MAX_TARGETS = 25; // one batch call holds at most 25 operations\nconst MAX_APPROVERS_NOTIFIED = 10;\nconst MAX_CREDIT_GRANT = 1_000_000;\nconst DAY_MS = 24 * 3_600_000;\nconst DEFAULT_DAILY_COUNT = 25;\nconst MAX_PER_ACCOUNT_PAGES = 5; // \u2264500 auto rows read for the per-account totals, else fail closed\n\nconst KINDS = ['credit_grant', 'plan_change', 'suspend', 'reactivate', 'data_fix'] as const;\ntype Kind = (typeof KINDS)[number];\n/** Only these kinds may auto-apply under a policy threshold (they carry an amount). */\nconst AUTO_KINDS: ReadonlySet<Kind> = new Set<Kind>(['credit_grant']);\n\ninterface Payload {\n kind?: string;\n account_ids?: unknown;\n amount?: unknown;\n payload?: unknown;\n reason?: unknown;\n}\ntype Env = HttpFunctionEnvelope<Payload>;\ninterface Policy {\n approver_permission?: string;\n auto_approve_below?: number | null;\n /** rolling-24 h cap on the SUM of auto-applied amounts, per requester and per account */\n auto_approve_daily_cap?: number | null;\n /** rolling-24 h cap on the NUMBER of auto-applied grants per requester (the race backstop) */\n auto_approve_daily_count?: number | null;\n expires_after_hours?: number | null;\n enabled?: boolean;\n}\ninterface BatchResult { op: string; status: number; data?: { item_id?: string }; error?: { code?: string; message?: string } }\ntype Patch = { $inc: { credits: number } } | { data: Record<string, unknown> };\n\nconst hdrs = (jwt: string) => ({ authorization: `Bearer ${jwt}`, 'content-type': 'application/json' });\nconst enc = (o: unknown) => encodeURIComponent(JSON.stringify(o));\nconst fail = (status: number, error: string, extra: Record<string, unknown> = {}) =>\n Response.json({ error, ...extra }, { status });\n\n/** The per-kind allow-list. Returns the account patch, or an error string. */\nfunction patchFor(kind: Kind, amount: number, payload: Record<string, unknown>): Patch | string {\n const keys = Object.keys(payload);\n const only = (allowed: string[]) => keys.filter((k) => !allowed.includes(k));\n switch (kind) {\n case 'credit_grant': {\n if (keys.length) return `credit_grant takes no payload (got ${keys.join(', ')})`;\n if (!Number.isInteger(amount) || amount < 1 || amount > MAX_CREDIT_GRANT) {\n return `credit_grant needs an integer amount between 1 and ${MAX_CREDIT_GRANT}`;\n }\n return { $inc: { credits: amount } };\n }\n case 'plan_change': {\n const extra = only(['plan']);\n if (extra.length) return `plan_change allows only { plan } (got ${extra.join(', ')})`;\n const plan = payload.plan;\n if (typeof plan !== 'string' || !/^[a-z0-9][a-z0-9_-]{0,63}$/.test(plan)) return 'plan_change needs { plan: \"<plan id>\" }';\n return { data: { plan } };\n }\n case 'suspend':\n case 'reactivate': {\n if (keys.length) return `${kind} takes no payload (got ${keys.join(', ')})`;\n return { data: { status: kind === 'suspend' ? 'suspended' : 'active' } };\n }\n case 'data_fix': {\n const extra = only(['name', 'email', 'seats']);\n if (extra.length) return `data_fix allows only name, email, seats (got ${extra.join(', ')})`;\n if (!keys.length) return 'data_fix needs at least one of name, email, seats';\n if ('name' in payload && (typeof payload.name !== 'string' || !payload.name.trim() || payload.name.length > 200)) return 'name must be a non-empty string';\n if ('email' in payload && (typeof payload.email !== 'string' || !/^[^\\s@]+@[^\\s@]+$/.test(payload.email))) return 'email must be an e-mail address';\n if ('seats' in payload && (!Number.isInteger(payload.seats) || (payload.seats as number) < 0)) return 'seats must be a non-negative integer';\n return { data: { ...payload } };\n }\n }\n}\n\n/** The staff org of the caller (their active membership) + a revocation-grade\n * permission check in it. Returns the org id, or the 403 to send. */\nasync function requirePermission(base: string, orgs: string, userId: string, permission: string): Promise<string | Response> {\n const c = await fetch(`${base}/v1/orgs/session-claims?user_id=${encodeURIComponent(userId)}`, { headers: hdrs(orgs) });\n if (!c.ok) return fail(502, 'orgs_unavailable', { status: c.status });\n const orgId = ((await c.json()) as { data?: { org_id?: string | null } }).data?.org_id;\n if (!orgId) return fail(403, 'forbidden', { permission, message: 'you are not a member of the staff organization' });\n const r = await fetch(\n `${base}/v1/orgs/${encodeURIComponent(orgId)}/check?user_id=${encodeURIComponent(userId)}&permission=${encodeURIComponent(permission)}`,\n { headers: hdrs(orgs) },\n );\n if (!r.ok) return fail(502, 'orgs_unavailable', { status: r.status });\n const allowed = ((await r.json()) as { data?: { allowed?: boolean } }).data?.allowed === true;\n return allowed ? orgId : fail(403, 'forbidden', { permission, message: `you need the '${permission}' permission` });\n}\n\nasync function loadPolicy(base: string, cms: string, kind: string): Promise<Policy> {\n const r = await fetch(`${base}/v1/cms/items/approval_policies?filter=${enc({ kind })}&limit=1`, { headers: hdrs(cms) });\n if (!r.ok) return {};\n const row = ((await r.json()) as { data?: { items?: Array<{ data?: Policy }> } }).data?.items?.[0];\n return row?.data ?? {};\n}\n\n/** Sum of the caller's auto-applied grants since `since` (one bounded aggregate). null = unreadable. */\nasync function autoGrantedByRequester(base: string, cms: string, me: string, since: string): Promise<number | null> {\n const r = await fetch(`${base}/v1/cms/items/change_requests/aggregate`, {\n method: 'POST',\n headers: hdrs(cms),\n body: JSON.stringify({\n aggregates: [{ fn: 'sum', field: 'amount', as: 'total' }],\n filter: { requested_by: me, decided_by: 'policy', kind: 'credit_grant' },\n window: { field: 'requested_at', since },\n }),\n });\n if (!r.ok) return null;\n const total = ((await r.json()) as { data?: { groups?: Array<{ total?: number | string | null }> } }).data?.groups?.[0]?.total;\n return Number(total ?? 0) || 0;\n}\n\n/** Auto-applied grant totals per target account since `since`. null = unreadable or too many rows. */\nasync function autoGrantedPerAccount(base: string, cms: string, accounts: string[], since: string): Promise<Map<string, number> | null> {\n const totals = new Map<string, number>();\n if (!accounts.length) return totals;\n const filter = { decided_by: 'policy', kind: 'credit_grant', account_ref: { $in: accounts }, requested_at: { $gte: since } };\n let cursor: string | null = null;\n for (let page = 0; page < MAX_PER_ACCOUNT_PAGES; page++) {\n const r = await fetch(\n `${base}/v1/cms/items/change_requests?filter=${enc(filter)}&limit=100${cursor ? `&cursor=${encodeURIComponent(cursor)}` : ''}`,\n { headers: hdrs(cms) },\n );\n if (!r.ok) return null;\n const body = ((await r.json()) as { data?: { items?: Array<{ data?: { account_ref?: string; amount?: number } }>; next_cursor?: string | null } }).data;\n for (const it of body?.items ?? []) {\n const a = it.data?.account_ref;\n if (a) totals.set(a, (totals.get(a) ?? 0) + (Number(it.data?.amount) || 0));\n }\n cursor = body?.next_cursor ?? null;\n if (!cursor) return totals;\n }\n return null;\n}\n\n/** Org members whose role grants `permission` (custom role rows or the owner's `*`). */\nasync function approversFor(base: string, orgs: string, orgId: string, permission: string): Promise<string[]> {\n const [m, r] = await Promise.all([\n fetch(`${base}/v1/orgs/${encodeURIComponent(orgId)}/members`, { headers: hdrs(orgs) }),\n fetch(`${base}/v1/orgs/roles`, { headers: hdrs(orgs) }),\n ]);\n if (!m.ok || !r.ok) return [];\n const members = ((await m.json()) as { data?: { members?: Array<{ user_id: string; role: string }> } }).data?.members ?? [];\n const roles = ((await r.json()) as { data?: { builtin?: RoleRow[]; custom?: RoleRow[] } }).data ?? {};\n const granting = new Set(\n [...(roles.builtin ?? []), ...(roles.custom ?? [])]\n .filter((x) => (x.permissions ?? []).includes(permission) || (x.permissions ?? []).includes('*'))\n .map((x) => x.role_key),\n );\n return members.filter((x) => granting.has(x.role)).map((x) => x.user_id);\n}\ninterface RoleRow { role_key: string; permissions?: string[] }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const { cms, orgs, notifications } = env.scoped_jwts ?? {};\n const me = env.end_user?.id;\n if (!me) return fail(401, 'sign_in_required', { message: 'call this with a signed-in staff session' });\n if (!cms || !orgs || !notifications) return fail(500, 'missing_scopes');\n\n // \u2500\u2500 validate the request BEFORE any permission or data read \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n const p = env.payload ?? {};\n const kind = p.kind as Kind;\n if (!KINDS.includes(kind)) return fail(422, 'invalid_payload', { message: `kind must be one of ${KINDS.join(', ')}` });\n const ids = Array.isArray(p.account_ids) ? [...new Set(p.account_ids)] : [];\n if (ids.length < 1 || ids.length > MAX_TARGETS || !ids.every((x) => typeof x === 'string' && /^[A-Za-z0-9_-]{1,64}$/.test(x))) {\n return fail(422, 'invalid_payload', { message: `account_ids must be 1\u2013${MAX_TARGETS} item ids` });\n }\n const reason = typeof p.reason === 'string' ? p.reason.trim() : '';\n if (!reason || reason.length > 2000) return fail(422, 'invalid_payload', { message: 'reason is required (\u22642000 chars)' });\n const amount = kind === 'credit_grant' ? Number(p.amount) : 0;\n if (p.payload !== undefined && (typeof p.payload !== 'object' || p.payload === null || Array.isArray(p.payload))) {\n return fail(422, 'invalid_payload', { message: 'payload must be an object' });\n }\n const patchPayload = (p.payload ?? {}) as Record<string, unknown>;\n const patch = patchFor(kind, amount, patchPayload);\n if (typeof patch === 'string') return fail(422, 'invalid_payload', { message: patch });\n\n const orgId = await requirePermission(base, orgs, me, 'changes.request');\n if (orgId instanceof Response) return orgId;\n\n // \u2500\u2500 the targets must exist (one round trip for up to 25 reads) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n const got = await fetch(`${base}/v1/cms/batch`, {\n method: 'POST',\n headers: hdrs(cms),\n body: JSON.stringify({ ops: ids.map((id) => ({ op: 'get', collection: 'accounts', id })) }),\n });\n if (!got.ok) return fail(502, 'cms_unavailable', { status: got.status });\n const reads = ((await got.json()) as { data?: { results?: BatchResult[] } }).data?.results ?? [];\n const found = ids.filter((_, i) => reads[i]?.status === 200) as string[];\n const results: Array<{ account_id: string; request_id?: string | undefined; state: string; error?: string; reason?: string }> =\n ids.filter((_, i) => reads[i]?.status !== 200).map((id) => ({ account_id: id as string, state: 'rejected', error: 'account_not_found' }));\n\n const policy = await loadPolicy(base, cms, kind);\n const now = new Date().toISOString();\n const row = (accountId: string) => ({\n kind, requested_by: me, amount, account_ref: accountId, requested_at: now,\n request_key: `${env.idempotency_key}:${accountId}`, payload: patchPayload, reason,\n });\n const autoEligible = AUTO_KINDS.has(kind) && policy.enabled !== false\n && typeof policy.auto_approve_below === 'number' && amount < policy.auto_approve_below;\n\n // Accounts that end up as pending requests: everything, unless auto-eligible\n // AND inside the rolling daily caps (then only the ones that crossed them).\n let toFile: string[] = found;\n const capped = new Set<string>();\n\n if (autoEligible) {\n // \u2500\u2500 the CUMULATIVE cap, before anything is applied \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n // A per-call threshold alone is bypassable by splitting: twenty grants\n // of 499 are 9,980 credits nobody approved. So the policy also caps the\n // SUM of auto-applied grants over a rolling 24 h, per requester AND per\n // account; a grant that would cross either cap is filed `pending` for\n // an approver instead. Unreadable totals fail closed (filed pending).\n const since = new Date(Date.now() - DAY_MS).toISOString();\n const dailyCap = typeof policy.auto_approve_daily_cap === 'number'\n ? policy.auto_approve_daily_cap\n : policy.auto_approve_below!; // no cap row \u2192 at most one threshold's worth a day\n const dailyCount = typeof policy.auto_approve_daily_count === 'number' && policy.auto_approve_daily_count >= 1\n ? policy.auto_approve_daily_count\n : DEFAULT_DAILY_COUNT;\n let mine = await autoGrantedByRequester(base, cms, me, since);\n const perAccount = await autoGrantedPerAccount(base, cms, found, since);\n toFile = [];\n for (const accountId of found) {\n const onAccount = perAccount === null ? null : (perAccount.get(accountId) ?? 0);\n if (mine === null || onAccount === null || mine + amount > dailyCap || onAccount + amount > dailyCap) {\n toFile.push(accountId);\n capped.add(accountId);\n continue;\n }\n // \u2500\u2500 under the threshold and the caps: request row + account, atomically.\n // The lock + guard are the concurrency backstop: the totals above are\n // a read, so two calls racing could both pass them. Under ONE lock per\n // requester, the guard admits at most `dailyCount` auto-applied grants\n // per requester per 24 h, so racing calls can overshoot the sum cap by\n // at most that many sub-threshold grants, never without bound.\n const r = await fetch(`${base}/v1/cms/batch`, {\n method: 'POST',\n headers: hdrs(cms),\n body: JSON.stringify({\n atomic: true,\n ops: [\n { op: 'create', collection: 'change_requests',\n data: { ...row(accountId), state: 'applied', auto: true, decided_by: 'policy', decided_at: now },\n lock: `auto-grant:${me}`,\n guards: [{\n filter: { requested_by: me, decided_by: 'policy', kind, requested_at: { $gte: since } },\n max: dailyCount,\n message: 'daily auto-approve count reached',\n }] },\n { op: 'patch', collection: 'accounts', id: accountId, ...patch },\n ],\n }),\n });\n const out = ((await r.json().catch(() => ({}))) as { data?: { ok?: boolean; results?: BatchResult[] } }).data;\n if (r.ok && out?.ok) {\n results.push({ account_id: accountId, request_id: out.results?.[0]?.data?.item_id, state: 'applied' });\n mine += amount;\n perAccount!.set(accountId, onAccount + amount);\n continue;\n }\n const bad = out?.results?.find((x) => x.status >= 300 && x.error?.code !== 'rolled_back' && x.error?.code !== 'not_run');\n const code = bad?.error?.code ?? `http_${r.status}`;\n if (code === 'guard_failed') { toFile.push(accountId); capped.add(accountId); continue; }\n results.push({ account_id: accountId, state: code === 'unique_violation' ? 'duplicate' : 'failed', error: code });\n }\n }\n\n // \u2500\u2500 file pending requests (\u226425 creates in one call) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n if (toFile.length) {\n const r = await fetch(`${base}/v1/cms/batch`, {\n method: 'POST',\n headers: hdrs(cms),\n body: JSON.stringify({\n ops: toFile.map((accountId) => ({ op: 'create', collection: 'change_requests', data: { ...row(accountId), state: 'pending', auto: false } })),\n }),\n });\n if (!r.ok) return fail(502, 'cms_unavailable', { status: r.status });\n const created = ((await r.json()) as { data?: { results?: BatchResult[] } }).data?.results ?? [];\n toFile.forEach((accountId, i) => {\n const c = created[i];\n if (c && c.status < 300) {\n results.push({ account_id: accountId, request_id: c.data?.item_id, state: 'pending',\n ...(capped.has(accountId) ? { reason: 'daily_auto_cap' } : {}) });\n } else results.push({ account_id: accountId, state: c?.error?.code === 'unique_violation' ? 'duplicate' : 'failed', error: c?.error?.code ?? 'not_run' });\n });\n }\n\n // \u2500\u2500 tell the people who can approve it (never the requester) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n const filed = results.filter((x) => x.state === 'pending');\n let notified = 0;\n if (filed.length) {\n const permission = policy.approver_permission || 'changes.approve';\n const approvers = (await approversFor(base, orgs, orgId, permission)).filter((u) => u !== me).slice(0, MAX_APPROVERS_NOTIFIED);\n const what = kind === 'credit_grant' ? `${kind} of ${amount} credits` : kind;\n for (const approver of approvers) {\n const s = await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: { ...hdrs(notifications), 'idempotency-key': `${env.idempotency_key}:notify:${approver}` },\n body: JSON.stringify({\n user_id: approver,\n template: 'transactional',\n channel: 'both',\n data: {\n subject: `Approval needed: ${what} (${filed.length} account${filed.length === 1 ? '' : 's'})`,\n paragraph: `Requested by ${me}. Reason: ${reason.slice(0, 300)}. Review it at ${CONSOLE_URL}/requests?state=pending`,\n },\n }),\n });\n if (s.ok) notified++;\n }\n }\n return Response.json({ results, notified }, { status: filed.length ? 201 : 200 });\n },\n};\n"
18293
+ }
18294
+ },
18253
18295
  {
18254
18296
  "id": "funnel-saas",
18255
18297
  "title": "Funnel SaaS (studio starter)",
@@ -18394,6 +18436,83 @@ export default defineConfig({
18394
18436
  "heartbeat.ts": "// heartbeat.ts \u2014 THE SILENT-PROVIDER DETECTOR (a vxil function, cron trigger).\n//\n// Trigger: cron `0 7 * * *`. For each provider you list in PROVIDERS:\n// 1. read the newest PROCESSED production webhook events \u2014\n// `GET /v1/payments/webhook-events?provider=<p>&environment=production\n// &outcome=processed&limit=100` (newest first),\n// 2. days_since = now \u2212 the newest event's received_at,\n// 3. threshold = max(FLOOR_DAYS, ALERT_MULTIPLE \xD7 the MEDIAN gap between\n// those events) \u2014 your own cadence, not a number somebody picked,\n// 4. ok | stale | broken (or could_not_check when the read itself failed),\n// 5. compare with the stored row for that provider and post to Slack ONLY on\n// a crossing; a return to ok posts RESOLVED.\n//\n// Why the median and not the mean: one migration backfill or one Black Friday\n// inflates a mean for months. The median is what \"normal\" actually looks like.\n//\n// Why `?environment=production`: sandbox traffic is developer noise. A provider\n// can be chatty in test mode while production has been silent for a month \u2014\n// that is precisely the outage this function exists to catch.\n//\n// Run it by hand any time: `vxil functions invoke heartbeat`. It performs the\n// real check and returns the per-provider verdict as JSON; it still only posts\n// if a state genuinely crossed.\n\n// \u2500\u2500 Tune these. They are the whole policy. \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n// cron-walk: single-read \u2014 a sample of the newest 100 events per provider is the measurement itself.\n\nimport type { CronFunctionEnvelope } from '@vxil/sdk';\n\n/** Providers to probe. Must be names the payments API accepts. Drop the ones\n * you do not use \u2014 a provider that has never sent an event is skipped anyway. */\nconst PROVIDERS = ['stripe', 'paddle', 'paypal', 'revenuecat'] as const;\n/** Alert once silence exceeds this multiple of your median inter-event gap. */\nconst ALERT_MULTIPLE = 3;\n/** \u2026but never sooner than this many days, however chatty your integration is.\n * A low-volume app can legitimately go a week without a single lifecycle event. */\nconst FLOOR_DAYS = 7;\n/** Past this multiple of the threshold, `stale` becomes `broken`. This split is\n * the blueprint's own choice \u2014 the SLO defines the alert threshold, not the\n * severity ladder \u2014 so move it wherever your escalation wants it. */\nconst BROKEN_MULTIPLE = 2;\n/** Gaps needed before a median means anything. Below this the floor is used. */\nconst MIN_GAPS_FOR_MEDIAN = 5;\n/** Events sampled per provider (the API caps a page at 100). */\nconst SAMPLE = 100;\n\ntype State = 'ok' | 'stale' | 'broken' | 'could_not_check';\n\ntype Envelope = CronFunctionEnvelope;\ninterface WebhookEvent { event_id: string; provider: string; received_at: string | null }\ninterface StateData {\n provider: string; state?: State; detail?: string;\n days_since?: number; threshold_days?: number;\n last_event_at?: string; checked_at?: string; last_event_id?: string;\n}\ninterface Item { item_id: string; version?: number; data: StateData }\n\ninterface Verdict {\n provider: string;\n state: State;\n detail: string;\n days_since: number | null;\n threshold_days: number | null;\n last_event_at: string | null;\n last_event_id: string | null;\n}\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Envelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const payments = env.scoped_jwts?.payments;\n const cms = env.scoped_jwts?.cms;\n const hook = env.secrets?.slack_webhook_url;\n if (!payments || !cms) return Response.json({ error: 'missing payments:read / cms scope' }, { status: 403 });\n if (!hook) return Response.json({ error: 'missing secret slack_webhook_url' }, { status: 409 });\n\n const store = new Store(base, cms);\n const checked: Verdict[] = [];\n const crossings: string[] = [];\n\n for (const provider of PROVIDERS) {\n const verdict = await check(base, payments, provider);\n if (!verdict) continue; // never sent an event \u2192 not part of this integration\n checked.push(verdict);\n const line = await reconcile(store, verdict);\n if (line) crossings.push(line);\n }\n\n let posted = 0;\n for (const text of crossings) {\n const r = await post(hook, text);\n if (r.ok) posted++;\n }\n return Response.json({ checked, crossings: crossings.length, posted });\n },\n};\n\n/** Measure one provider. `null` = the provider has never delivered a processed\n * production event, so there is no cadence to be silent against \u2014 the SLO's\n * \"once the tenant has ever received one\" precondition. */\nasync function check(base: string, jwt: string, provider: string): Promise<Verdict | null> {\n const url = `${base}/v1/payments/webhook-events`\n + `?provider=${encodeURIComponent(provider)}&environment=production&outcome=processed&limit=${SAMPLE}`;\n const res = await fetch(url, { headers: { authorization: `Bearer ${jwt}` } }).catch(() => null);\n if (!res) {\n return verdict(provider, 'could_not_check', `could not reach the payments event log for ${provider}`);\n }\n if (!res.ok) {\n // 501 capability_not_enabled / 403 missing scope are configuration, not silence.\n return verdict(provider, 'could_not_check', `payments event log returned ${res.status} for ${provider}`);\n }\n const body = (await res.json().catch(() => ({}))) as { data?: { events?: WebhookEvent[] } };\n const events = (body.data?.events ?? []).filter((e) => e.received_at);\n if (events.length === 0) return null;\n\n // The API orders by received_at DESC, so [0] is the newest.\n const times = events.map((e) => Date.parse(e.received_at!)).filter(Number.isFinite).sort((a, b) => b - a);\n if (times.length === 0) return null;\n const last = times[0]!;\n const daysSince = round2((Date.now() - last) / 86_400_000);\n\n const gaps: number[] = [];\n for (let i = 0; i + 1 < times.length; i++) gaps.push((times[i]! - times[i + 1]!) / 86_400_000);\n const medianGap = gaps.length >= MIN_GAPS_FOR_MEDIAN ? median(gaps) : null;\n const threshold = round2(Math.max(FLOOR_DAYS, medianGap === null ? 0 : ALERT_MULTIPLE * medianGap));\n\n const state: State = daysSince <= threshold ? 'ok'\n : daysSince <= threshold * BROKEN_MULTIPLE ? 'stale'\n : 'broken';\n\n const cadence = medianGap === null\n ? `only ${gaps.length} gap(s) sampled \u2014 using the ${FLOOR_DAYS}-day floor`\n : `median gap ${round2(medianGap)}d \xD7 ${ALERT_MULTIPLE}, floor ${FLOOR_DAYS}d`;\n const detail = state === 'ok'\n ? `${provider}: last processed event ${daysSince}d ago (threshold ${threshold}d \u2014 ${cadence})`\n : `${provider} has sent no processed production event for ${daysSince}d \u2014 expected one within ${threshold}d (${cadence})`;\n\n return {\n provider, state, detail,\n days_since: daysSince,\n threshold_days: threshold,\n last_event_at: new Date(last).toISOString(),\n last_event_id: events[0]!.event_id ?? null,\n };\n}\n\n/** Persist the verdict; return a message ONLY when the state crossed. */\nasync function reconcile(store: Store, v: Verdict): Promise<string | null> {\n const now = new Date().toISOString();\n const data: StateData = {\n provider: v.provider, state: v.state, detail: v.detail, checked_at: now,\n ...(v.days_since !== null ? { days_since: v.days_since } : {}),\n ...(v.threshold_days !== null ? { threshold_days: v.threshold_days } : {}),\n ...(v.last_event_at ? { last_event_at: v.last_event_at } : {}),\n ...(v.last_event_id ? { last_event_id: v.last_event_id } : {}),\n };\n\n const current = await store.byProvider(v.provider);\n if (!current) {\n // First run. A healthy first sighting is remembered silently; an unhealthy\n // one is worth saying out loud immediately \u2014 you were already in the outage.\n const created = await store.create(data);\n return created && v.state !== 'ok' ? format(v) : null; // 409 \u21D2 a racing tick owns it\n }\n\n const prev = current.data.state ?? 'ok';\n if (prev === v.state) {\n // No crossing: refresh the measurement, stay quiet. A permanently-broken\n // provider therefore costs exactly one message, not one per day.\n await store.patch(current.item_id, current.version, data);\n return null;\n }\n // A CROSSING. If-Match makes exactly one of two overlapping ticks the winner.\n const ok = await store.patch(current.item_id, current.version, data);\n if (!ok) return null;\n return v.state === 'ok'\n ? `\u{1F7E2} RESOLVED: ${v.provider} is delivering again \u2014 last processed event ${v.days_since}d ago (was ${prev})`\n : format(v);\n}\n\nfunction format(v: Verdict): string {\n const dot = v.state === 'broken' ? '\u{1F534}' : v.state === 'stale' ? '\u{1F7E0}' : '\u{1F535}';\n const label = v.state === 'could_not_check' ? 'CHECK FAILED' : v.state.toUpperCase();\n return `${dot} [${label}] payments heartbeat \u2014 ${v.detail}`;\n}\n\n// \u2500\u2500 the cms state store (the REST envelope: { data: { items: [{item_id, version, data}] } }) \u2500\u2500\nclass Store {\n constructor(private base: string, private jwt: string) {}\n private h() { return { authorization: `Bearer ${this.jwt}`, 'content-type': 'application/json' }; }\n async byProvider(provider: string): Promise<Item | null> {\n const filter = encodeURIComponent(JSON.stringify({ provider }));\n const res = await fetch(`${this.base}/v1/cms/items/heartbeat_state?filter=${filter}&limit=1`, { headers: this.h() });\n if (!res.ok) return null;\n const body = (await res.json()) as { data?: { items?: Item[] } };\n return body.data?.items?.[0] ?? null;\n }\n /** false on 409 \u2014 `provider` is unique, so a concurrent tick already claimed it. */\n async create(data: StateData): Promise<boolean> {\n const res = await fetch(`${this.base}/v1/cms/items/heartbeat_state`, {\n method: 'POST', headers: this.h(), body: JSON.stringify({ status: 'published', data }),\n });\n return res.ok;\n }\n /** false on 409 version_conflict \u2014 another tick transitioned this provider first. */\n async patch(itemId: string, version: number | undefined, data: StateData): Promise<boolean> {\n const res = await fetch(`${this.base}/v1/cms/items/heartbeat_state/${itemId}`, {\n method: 'PATCH',\n headers: { ...this.h(), ...(version !== undefined ? { 'if-match': String(version) } : {}) },\n body: JSON.stringify({ data }),\n });\n return res.ok;\n }\n}\n\n// \u2500\u2500 the chat webhook: `text` is Slack's field, `content` is Discord's \u2014 send both,\n// EXCEPT to a Google Chat space webhook (chat.googleapis.com), which takes\n// `{ text }` alone \u2014 an unknown field can be rejected there, so it is dropped.\n// (Duplicated from templates/alerts-to-slack on purpose: a blueprint is a\n// self-contained clone, and every file under functions/ must be a declared\n// entry point, so there is no place for a shared module to live.)\nasync function post(url: string, text: string): Promise<{ ok: boolean; status: number }> {\n let host = '';\n try { host = new URL(url).hostname; } catch { /* unparseable \u2192 the generic body below */ }\n const body = host === 'chat.googleapis.com' ? { text } : { text, content: text };\n const res = await fetch(url, {\n method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body),\n }).catch(() => null);\n return { ok: Boolean(res?.ok), status: res?.status ?? 0 };\n}\n\nfunction verdict(provider: string, state: State, detail: string): Verdict {\n return { provider, state, detail, days_since: null, threshold_days: null, last_event_at: null, last_event_id: null };\n}\nfunction median(xs: number[]): number {\n const s = [...xs].sort((a, b) => a - b);\n const mid = s.length >> 1;\n return s.length % 2 ? s[mid]! : (s[mid - 1]! + s[mid]!) / 2;\n}\nconst round2 = (n: number) => Math.round(n * 100) / 100;\n"
18395
18437
  }
18396
18438
  },
18439
+ {
18440
+ "id": "ops-dashboard",
18441
+ "title": "Ops dashboard",
18442
+ "vertical": "ops",
18443
+ "summary": "A private team dashboard over your own services and stores \u2014 a cron function polls your payments provider, any REST API and your other vxil projects into live KPI tiles and hourly/daily trends, received webhooks and payments events land in one updates feed, threshold rules open incidents once per crossing with a chat and e-mail alert, revenue-grade numbers are readable only by the roles you name, and staff sign in through your identity provider only.",
18444
+ "collections": [
18445
+ "integrations",
18446
+ "metric_points",
18447
+ "metric_latest",
18448
+ "events_feed",
18449
+ "alert_rules",
18450
+ "incidents",
18451
+ "ops_state"
18452
+ ],
18453
+ "features": [
18454
+ "auth",
18455
+ "orgs",
18456
+ "cms",
18457
+ "realtime",
18458
+ "notifications",
18459
+ "webhooks",
18460
+ "functions"
18461
+ ],
18462
+ "hasFunctions": true,
18463
+ "byoKeys": [
18464
+ "stripe_key",
18465
+ "shop_token",
18466
+ "vxil_peer_key",
18467
+ "vxil_read_key",
18468
+ "slack_webhook_url",
18469
+ "oidc_client_secret"
18470
+ ],
18471
+ "configSrc": "import { defineConfig } from '@vxil/config';\n\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n// \"Ops dashboard\" \u2014 a PRIVATE dashboard for one team, wired to the team's own\n// services and stores: a payments provider, a shop or any REST API, the team's\n// other vxil projects, and inbound webhooks. Live KPI tiles, hourly and daily\n// trends, one updates feed, threshold alerts that open incidents, team roles.\n//\n// DISTRIBUTION OVER SHIPPED PRIMITIVES \u2014 there is no connectors feature: a\n// connector is a function + a secret + egressAllow. There is no dashboard\n// engine either: the tiles are cms rows, the trends are cms rows, the feed is\n// cms rows, and your own frontend renders them (examples/apps/team-dashboard).\n//\n// pull \u2192 functions `collect` polls each integration on a cron and writes\n// `metric_latest` (one row per series) + one `metric_points` row\n// per CLOSED hour; `drain-inbound` turns received webhooks into\n// `events_feed` rows\n// push \u2192 `drain-inbound` also binds `payments.` platform events, so the\n// payments integration's lifecycle lands in the feed in seconds\n// evaluate \u2192 `evaluate-alerts` checks `alert_rules` against `metric_latest`\n// and opens/resolves `incidents` once per crossing (+ chat + mail)\n// show \u2192 realtime change-data on three channels; staff read everything\n// with a read-only browser key; every staff WRITE goes through\n// `ops-action`, which checks the caller's org permission first\n//\n// Plan cadence: as shipped this is a Developer-plan config (4 crons, the\n// fastest every minute). The Free plan allows 5 functions and 3 crons, the\n// fastest every 15 minutes: set the two 5-minute schedules to '*/15 * * * *'\n// and drop drain-inbound's cron trigger (its `payments.` trigger still works),\n// or fold compact-daily into evaluate-alerts. See the README's limits table.\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n/** Org roles allowed to READ the restricted (revenue-grade) numbers. `owner` is\n * the built-in god role; `finance` and `ops-admin` are custom roles you define\n * once over the API (README step 3). An identity-provider group with the same\n * name works too: the IdP's `groups` claim and the org role travel together. */\nconst MONEY_READERS = ['finance', 'ops-admin', 'owner'];\n\nexport default defineConfig({\n env: 'staging',\n\n features: {\n // \u2500\u2500 Staff-only sign-in \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n // OIDC-ONLY + allowedDomains IS THE STAFF GATE. Never turn magic link or\n // email sign-up back on in a project holding company data: the collections\n // below are shared (no owner field) and every signed-in person reads all of\n // them, so an open sign-up would let ANY address read every number here.\n auth: {\n methods: { emailPassword: false, magicLink: false },\n providers: {\n oidc: {\n issuer: 'https://login.example-idp.com', // your identity provider (https, no query)\n clientId: 'ops-dashboard', // not a secret \u2014 it rides every authorize URL\n clientSecretRef: 'secret:oidc_client_secret',\n claims: { email: 'email', name: 'name', roles: 'groups' }, // IdP groups \u2192 session roles\n allowedDomains: ['example.com'], // fail-closed: any other domain is refused\n autoLink: true,\n },\n },\n // The member's org role rides the session, so `readRoles` gates without a\n // round-trip. It is a snapshot: revocation-grade checks call orgs (the\n // functions do, before every write).\n orgClaims: { enabled: true },\n session: { ttlMinutes: 60, refreshTtlDays: 7 },\n security: { allowedRedirectOrigins: ['https://ops.example.com', 'http://localhost:5173'] },\n },\n\n // Roles are rows, not config \u2014 the README defines ops-viewer, operator,\n // finance and ops-admin with `POST /v1/orgs/roles` and creates the one org\n // (slug `ops`) every staff member joins.\n orgs: { enabled: true },\n\n realtime: {},\n\n // One alert channel besides chat: an e-mail + in-app message to every org\n // member holding operator, ops-admin or owner, through the built-in\n // `transactional` template (subject / paragraph / call-to-action). The mock\n // provider delivers nothing; switch to your mail provider before relying on it.\n notifications: { provider: 'mock', fromEmail: 'ops@ops-dashboard.example', inboxEnabled: true },\n\n // Inbound webhook sources (one per external service that pushes to you)\n // are runtime rows: `POST /v1/webhooks/sources` \u2014 see the README.\n webhooks: {},\n\n functions: { enabled: true },\n\n cms: {\n // OFF on purpose: these collections have no owner \u2014 staff read all of\n // them \u2014 and the browser key holds NO cms:write, so an end-user session\n // can read but never write. Every write is a function.\n strictEndUserScope: false,\n // An hourly series is 8,760 rows a year; compact-daily prunes hour rows\n // after 90 days. Raise this if you track more than ~150 series.\n limits: { maxItemsPerCollection: 200_000 },\n\n hooks: {\n // The time bucket is derived from the point's own timestamp \u2014 the\n // writer cannot disagree with it. Hour: '2026-10-04T13'; day: '2026-10-04'.\n point_bucket: {\n collection: 'metric_points',\n event: 'beforeWrite',\n kind: 'derive',\n field: 'bucket',\n expr: \"item.grain == 'day' ? substr(item.ts, 0, 10) : substr(item.ts, 0, 13)\",\n },\n // One row per (series, grain, bucket): the unique point_key makes a\n // re-delivered cron tick a clean 409 instead of a duplicate point.\n // Derives run in declaration order, so this sees the bucket above.\n point_key: {\n collection: 'metric_points',\n event: 'beforeWrite',\n kind: 'derive',\n field: 'point_key',\n expr: \"concat(item.series, '|', item.grain, '|', item.bucket)\",\n },\n // Incidents move forward only: open \u2192 acked \u2192 resolved (or open \u2192 resolved).\n incident_stage: {\n collection: 'incidents',\n event: 'beforeUpdate',\n kind: 'validate',\n expr:\n 'item.state == before.state'\n + \" || (before.state == 'open' && (item.state == 'acked' || item.state == 'resolved'))\"\n + \" || (before.state == 'acked' && item.state == 'resolved')\",\n message: 'incidents move open \u2192 acked \u2192 resolved',\n },\n rule_op: {\n collection: 'alert_rules',\n event: 'beforeWrite',\n kind: 'validate',\n expr: \"item.op == '>' || item.op == '<' || item.op == '>=' || item.op == '<=' || item.op == 'stale'\",\n message: \"op must be one of > < >= <= stale\",\n },\n },\n\n // Realtime change-data. A FULL frame carries every field of the row to\n // every subscriber \u2014 including a `readRoles` field the subscriber may not\n // read \u2014 so a collection with any readRoles field publishes IDS ONLY and\n // the client re-reads the row (the read applies the gate). The CI gate\n // `cdc-gated-fields` refuses a full frame on such a collection.\n cdc: {\n tiles_live: { collection: 'metric_latest', channel: 'ops:metrics', events: ['created', 'updated'], payload: 'ids' },\n incidents_live: { collection: 'incidents', channel: 'ops:incidents', payload: 'ids' },\n feed_live: { collection: 'events_feed', channel: 'ops:events', events: ['created'], payload: 'full' },\n },\n\n // The DECLARATIVE twin of functions/compact-daily.ts, shown and left off:\n //\n // readModels: {\n // daily_by_series: {\n // collection: 'metric_points', kind: 'aggregate',\n // spec: {\n // aggregates: [{ fn: 'avg', field: 'value', as: 'avg' }, { fn: 'max', field: 'value', as: 'max' }, { fn: 'count' }],\n // groupBy: ['series', 'bucket'], filter: { grain: 'hour' }, window: { field: 'ts', sinceDays: 2 },\n // },\n // // materialize: { cron: '15 0 * * *', to: 'metric_daily' },\n // },\n // },\n //\n // A materialized read model rewrites a rollup collection on a schedule.\n // This blueprint keeps the imperative function instead because it also\n // compacts the gated restricted_value, writes day rows into the SAME\n // collection the charts already read, and prunes old hour rows.\n },\n },\n\n // \u2500\u2500 Schema-as-code (\u22648 index slots per collection: s1\u2013s4 / n1\u2013n2 / t1\u2013t2) \u2500\u2500\n cms: {\n collections: {\n // One row per connected source. NON-SECRET settings only \u2014 a token never\n // lives here (ops-action strips token-looking keys from every write).\n integrations: {\n singular: 'integration',\n fields: {\n key: { type: 'string', required: true, unique: true, indexSlot: 's1' }, // e.g. stripe, shop, billing-project\n kind: { type: 'string', required: true, indexSlot: 's2', validation: { enum: ['stripe', 'rest-json', 'vxil-project', 'webhook-source'] } },\n status: { type: 'string', indexSlot: 's3', validation: { enum: ['ok', 'stale', 'broken', 'could_not_check', 'paused'] } },\n owner: { type: 'string', indexSlot: 's4' }, // the person to ask\n consecutive_failures: { type: 'int', indexSlot: 'n1', validation: { min: 0 } },\n poll_every_min: { type: 'int', indexSlot: 'n2', validation: { min: 1, max: 1440 } },\n last_ok_at: { type: 'datetime', indexSlot: 't1' },\n next_due_at: { type: 'datetime', indexSlot: 't2' }, // collect reads `next_due_at <= now`, sorted\n label: { type: 'string', validation: { max: 120 } },\n enabled: { type: 'bool', indexed: true },\n // base_url, path, series[] (name, json_path, unit, sensitive), source_id,\n // aggregates[], link_url \u2014 never a credential\n config: { type: 'json' },\n last_error: { type: 'text', validation: { max: 500 } },\n last_error_at: { type: 'datetime' },\n cursor: { type: 'string' }, // webhook sources: the newest event already in the feed\n },\n },\n\n // The trend store: one row per series per CLOSED hour (grain 'hour') and\n // one per day (grain 'day', written by compact-daily). Never raw samples.\n metric_points: {\n singular: 'metric_point',\n fields: {\n series: { type: 'string', required: true, indexSlot: 's1' }, // 'source:metric[:dim]'\n grain: { type: 'string', required: true, indexSlot: 's2', validation: { enum: ['hour', 'day'] } },\n bucket: { type: 'string', indexSlot: 's3' }, // derived (point_bucket)\n point_key: { type: 'string', unique: true, indexSlot: 's4' }, // derived (point_key)\n value: { type: 'float', indexSlot: 'n1' },\n restricted_value: { type: 'float', indexSlot: 'n2', readRoles: MONEY_READERS },\n ts: { type: 'datetime', required: true, indexSlot: 't1' }, // the bucket's start\n unit: { type: 'string' },\n samples: { type: 'int' },\n min: { type: 'float' },\n max: { type: 'float' },\n },\n },\n\n // The tiles: one row per series, patched on every collect tick.\n metric_latest: {\n singular: 'metric_latest',\n fields: {\n series: { type: 'string', required: true, unique: true, indexSlot: 's1' },\n source: { type: 'string', indexSlot: 's2' }, // the integration key\n status: { type: 'string', indexSlot: 's3', validation: { enum: ['ok', 'warn', 'crit', 'stale'] } },\n last_bucket: { type: 'string', indexSlot: 's4' }, // the newest hour already in metric_points\n value: { type: 'float', indexSlot: 'n1' },\n restricted_value: { type: 'float', indexSlot: 'n2', readRoles: MONEY_READERS },\n updated_at: { type: 'datetime', indexSlot: 't1' },\n label: { type: 'string' },\n unit: { type: 'string' },\n sensitive: { type: 'bool' }, // true \u21D2 the number lives in restricted_value only\n hour_samples: { type: 'int' },\n hour_min: { type: 'float' }, // non-sensitive series only\n hour_max: { type: 'float' },\n },\n },\n\n // The updates feed. Title + summary only \u2014 never a vendor payload, which\n // can carry personal data, and this collection's change-data is full.\n events_feed: {\n singular: 'event',\n fields: {\n source: { type: 'string', required: true, indexSlot: 's1' },\n kind: { type: 'string', indexSlot: 's2' },\n severity: { type: 'string', indexSlot: 's3', validation: { enum: ['info', 'warn', 'error'] } },\n ext_id: { type: 'string', unique: true, indexSlot: 's4' }, // the sender's event id \u21D2 a redelivery is a 409\n occurred_at: { type: 'datetime', indexSlot: 't1' },\n title: { type: 'string', validation: { max: 200 } },\n url: { type: 'string', validation: { max: 500 } },\n summary: { type: 'text', validation: { max: 2000 } },\n },\n },\n\n alert_rules: {\n singular: 'alert_rule',\n fields: {\n key: { type: 'string', required: true, unique: true, indexSlot: 's1' },\n series: { type: 'string', required: true, indexSlot: 's2' },\n state: { type: 'string', indexSlot: 's3', validation: { enum: ['ok', 'firing'] } },\n severity: { type: 'string', indexSlot: 's4', validation: { enum: ['info', 'warn', 'crit'] } },\n threshold: { type: 'float', indexSlot: 'n1' }, // for 'stale': minutes without a fresh value\n for_minutes: { type: 'int', indexSlot: 'n2', validation: { min: 0 } },\n last_eval_at: { type: 'datetime', indexSlot: 't1' },\n last_fired_at: { type: 'datetime', indexSlot: 't2' },\n op: { type: 'string', required: true }, // > < >= <= stale (rule_op hook)\n enabled: { type: 'bool', indexed: true },\n channels: { type: 'json' }, // { chat: bool, email: bool }\n breach_since: { type: 'datetime' },\n repeat_after_hours: { type: 'int', validation: { min: 0 } },\n title: { type: 'string', validation: { max: 200 } },\n },\n },\n\n incidents: {\n singular: 'incident',\n fields: {\n rule: { type: 'string', required: true, indexSlot: 's1' },\n state: { type: 'string', required: true, indexSlot: 's2', validation: { enum: ['open', 'acked', 'resolved'] } },\n severity: { type: 'string', indexSlot: 's3' },\n acked_by: { type: 'string', indexSlot: 's4' },\n opened_at: { type: 'datetime', indexSlot: 't1' },\n resolved_at: { type: 'datetime', indexSlot: 't2' },\n title: { type: 'string', validation: { max: 200 } },\n last_value: { type: 'float' }, // null for a sensitive series\n note: { type: 'text', validation: { max: 2000 } },\n },\n },\n\n // Function memory: the `__compact__` row holds the last day compact-daily finished.\n ops_state: {\n singular: 'ops_state',\n fields: {\n key: { type: 'string', required: true, unique: true, indexSlot: 's1' },\n updated_at: { type: 'datetime', indexSlot: 't1' },\n cursor: { type: 'string' },\n },\n },\n },\n },\n\n // \u2500\u2500 The code that isn't config (guide ch. 8) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n functions: {\n // PULL. Every 5 minutes: poll each due integration, write the tiles and the\n // closed hour. Also an http door \u2014 the \"Sync now\" button \u2014 that checks the\n // caller holds `integrations.sync` in the ops org first.\n collect: {\n entry: './functions/collect.ts',\n trigger: { kind: 'cron', schedule: '*/5 * * * *' }, // Free: '*/15 * * * *'\n triggers: [{ kind: 'http' }],\n scopes: ['cms:read', 'cms:write', 'orgs:read'],\n // Every ref here must hold a value or the function refuses to run\n // (409 secret_missing): drop the sources you do not poll.\n secrets: ['secret:stripe_key', 'secret:shop_token', 'secret:vxil_peer_key'],\n // Deny-by-default egress: every REST host you poll goes here, then\n // `vxil push`. www.githubstatus.com is the keyless demo source. Which\n // host each SECRET may go to is fixed in functions/collect.ts\n // (SECRET_HOST / PEER_SECRETS), not by an integration row.\n egressAllow: ['api.stripe.com', 'shop.example.com', 'www.githubstatus.com'],\n limits: { timeoutMs: 20_000 }, // one slow vendor call fails in 20 s, not 30\n },\n\n // PUSH. Every minute: received webhooks \u2192 feed rows. And within seconds:\n // this project's own payments-integration events (the `payments.` trigger).\n 'drain-inbound': {\n entry: './functions/drain-inbound.ts',\n trigger: { kind: 'cron', schedule: '* * * * *' }, // Free: remove this trigger (3 crons max)\n triggers: [{ kind: 'webhook', source: 'payments.' }],\n scopes: ['cms:read', 'cms:write'],\n // Received webhook events are not on a function's scoped callback, so the\n // drain reads them with a key of THIS project holding only webhooks:read.\n secrets: ['secret:vxil_read_key'],\n egressAllow: [],\n },\n\n // EVALUATE. Every 5 minutes: rules \xD7 latest values \u2192 incidents, once per crossing.\n 'evaluate-alerts': {\n entry: './functions/evaluate-alerts.ts',\n trigger: { kind: 'cron', schedule: '*/5 * * * *' }, // Free: '*/15 * * * *'\n scopes: ['cms:read', 'cms:write', 'notifications:send', 'orgs:read'],\n secrets: ['secret:slack_webhook_url'],\n // The chat host. Add 'discord.com' for a Discord webhook or\n // 'chat.googleapis.com' for a Google Chat space webhook.\n egressAllow: ['hooks.slack.com'],\n },\n\n // Nightly: hour rows \u2192 one day row per series, then prune old hour rows.\n 'compact-daily': {\n entry: './functions/compact-daily.ts',\n trigger: { kind: 'cron', schedule: '15 0 * * *' },\n scopes: ['cms:read', 'cms:write'],\n egressAllow: [],\n },\n\n // SHOW (the write half). The dashboard's only write door: ack/resolve an\n // incident, edit or toggle a rule, edit or pause an integration \u2014 each\n // gated on an org permission. Audited as the project; the person is\n // recorded in the row (acked_by).\n 'ops-action': {\n entry: './functions/ops-action.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write', 'orgs:read'],\n egressAllow: [],\n },\n },\n\n // References only \u2014 values are set once with `vxil secrets set` and never appear here.\n secrets: {\n stripe_key: { feature: 'functions', description: 'Stripe RESTRICTED key, read-only on balance and charges' },\n shop_token: { feature: 'functions', description: 'bearer token of the REST API collect polls (rest-json integrations)' },\n vxil_peer_key: { feature: 'functions', description: 'API key of ANOTHER vxil project: usage:read (+ cms:read for aggregates)' },\n vxil_read_key: { feature: 'functions', description: 'API key of THIS project holding only webhooks:read' },\n slack_webhook_url: { feature: 'functions', description: 'Slack incoming webhook (or Discord / Google Chat space webhook) URL' },\n oidc_client_secret: { feature: 'auth', description: 'OIDC client secret of the staff identity provider' },\n },\n\n seed: {\n cms: [\n {\n collection: 'integrations',\n items: [\n {\n key: 'demo', kind: 'rest-json', label: 'Public status page (demo)', status: 'ok', enabled: true,\n poll_every_min: 5, consecutive_failures: 0, next_due_at: '2026-01-01T00:00:00Z',\n config: {\n base_url: 'https://www.githubstatus.com', path: '/api/v2/incidents/unresolved.json', auth: 'none',\n series: [{ name: 'open_incidents', json_path: 'incidents.length', unit: 'count' }],\n },\n },\n {\n key: 'stripe', kind: 'stripe', label: 'Payments provider', status: 'paused', enabled: false,\n poll_every_min: 5, consecutive_failures: 0, config: {},\n },\n {\n key: 'shop', kind: 'rest-json', label: 'Shop API', status: 'paused', enabled: false,\n poll_every_min: 15, consecutive_failures: 0,\n config: {\n base_url: 'https://shop.example.com', path: '/api/stats',\n series: [\n { name: 'orders_1h', json_path: 'orders.last_hour', unit: 'count' },\n { name: 'revenue_1h', json_path: 'revenue.last_hour', unit: 'usd', sensitive: true },\n ],\n },\n },\n {\n key: 'billing-project', kind: 'vxil-project', label: 'Billing backend (another vxil project)', status: 'paused',\n enabled: false, poll_every_min: 60, consecutive_failures: 0,\n config: { aggregates: [{ name: 'signups_24h', collection: 'accounts', fn: 'count', window_hours: 24 }] },\n },\n {\n key: 'support-inbox', kind: 'webhook-source', label: 'Support tool webhooks', status: 'paused', enabled: false,\n poll_every_min: 1, consecutive_failures: 0, config: { source_id: 'whs_replace_me' },\n },\n ],\n },\n {\n collection: 'alert_rules',\n items: [\n {\n key: 'payment-error-rate', series: 'stripe:charge_error_rate_1h', op: '>', threshold: 5, for_minutes: 10,\n severity: 'crit', state: 'ok', enabled: true, repeat_after_hours: 4, channels: { chat: true, email: true },\n title: 'Payment error rate above 5%',\n },\n {\n key: 'orders-dry', series: 'shop:orders_1h', op: '<', threshold: 1, for_minutes: 60,\n severity: 'warn', state: 'ok', enabled: true, repeat_after_hours: 24, channels: { chat: true, email: false },\n title: 'No orders in the last hour',\n },\n {\n key: 'demo-stale', series: 'demo:open_incidents', op: 'stale', threshold: 30, for_minutes: 0,\n severity: 'warn', state: 'ok', enabled: true, repeat_after_hours: 24, channels: { chat: true, email: false },\n title: 'Status data older than 30 minutes',\n },\n ],\n },\n ],\n },\n});\n",
18472
+ "readme": '# Ops dashboard (ops)\n\nA **private dashboard for one team**, wired to the team\'s own services and stores: your payments provider, a\nshop or any REST API, your other vxil projects, and the webhooks those services send you. It shows **live KPI\ntiles**, **hourly and daily trends**, **one updates feed**, **threshold alerts that open incidents** (once per\ncrossing, with a chat and an e-mail message), and **team roles** \u2014 revenue-grade numbers are readable only by the\nroles you name. Staff sign in through your identity provider and nothing else.\n\nIt is distribution over shipped building blocks. There is **no connectors feature**: a connector is a function +\na secret + a host in `egressAllow`. There is no dashboard engine either: every tile, point, feed entry and\nincident is a cms row, and your own frontend renders them \u2014 the reference frontend is\n`examples/apps/team-dashboard`.\n\nWhat it teaches that no other blueprint does: **time series on the cms** (closed-hour writes, a derived bucket,\nnightly compaction), **gated numbers** (`readRoles` + change-data that never carries them), and a **read-only\nbrowser** whose every write goes through one permission-checked function.\n\n## 1. The shape\n\n```\n your services \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500 vxil project (this blueprint) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500 your frontend\n (examples/apps/team-dashboard)\n payments provider \u2500\u2510 PULL \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510 metric_latest \u2500\u2500 tiles \u2500\u2500\u2510\n REST / shop API \u2500\u2500\u253C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u25B6\u2502 collect \u2502 metric_points \u2500\u2500 trends \u2500\u2524 \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510\n other vxil project \u2518 (cron) \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u251C\u2500\u2500\u25B6\u2502 read-only key \u2502\n \u25B2 compact-daily (nightly) \u2502 \u2502 + staff session \u2502\n inbound webhooks \u2500\u2500\u2510 PUSH \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510 \u2502 \u2502 + realtime \u2502\n payments events \u2500\u2500\u2534\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u25B6\u2502 drain-inbound\u2502 events_feed \u2500\u2500 feed \u2500\u2500\u2524 \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u252C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518\n \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u2502 \u2502 every write\n EVALUATE \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510 incidents \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u25BC\n alert_rules \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u25B6\u2502 evaluate-alerts\u2502\u2500\u2500\u25B6 chat + e-mail \u250C\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2510\n \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 \u2502 ops-action \u2502 (org permission\n \u2514\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518 checked first)\n```\n\nThe four arrows:\n\n| Arrow | Who | How |\n|---|---|---|\n| **Pull** | `collect`, every 5 minutes | polls each due integration, patches `metric_latest`, writes one `metric_points` row per closed hour |\n| **Push** | `drain-inbound`, every minute + on `payments.` events | received webhooks and the payments integration\'s events \u2192 `events_feed` |\n| **Evaluate** | `evaluate-alerts`, every 5 minutes | `alert_rules` \xD7 `metric_latest` \u2192 `incidents`, chat, e-mail \u2014 once per crossing |\n| **Show** | realtime change-data + your frontend | staff read every collection; the only write door is `ops-action` |\n\n## 2. Apply it\n\n```bash\nmkdir ops-dashboard && cd ops-dashboard && vxil init --template ops-dashboard # init scaffolds into the current directory\nvxil quickstart --env staging --no-push # or `vxil link <slug> --env staging` for an existing project\nvxil projects workload <slug> staging # Free: functions deploy to staging/development projects\n# the secrets (values never touch the config) \u2014 every ref a function declares must hold a value (below):\nprintf \'%s\' "$OIDC_SECRET" | vxil secrets set auth/oidc_client_secret\nprintf \'%s\' "$STRIPE_RK" | vxil secrets set functions/stripe_key # a RESTRICTED, read-only key\nprintf \'%s\' "$SHOP_TOKEN" | vxil secrets set functions/shop_token\nprintf \'%s\' "$PEER_KEY" | vxil secrets set functions/vxil_peer_key # a key of ANOTHER project\nprintf \'%s\' "$READ_KEY" | vxil secrets set functions/vxil_read_key # a key of THIS project: webhooks:read\nprintf \'%s\' "$SLACK_URL" | vxil secrets set functions/slack_webhook_url\nvxil push # collections, hooks, change-data, the five functions\nvxil seed # the demo integration + three example rules\nvxil functions invoke collect # one tick by hand: the demo source fills its first tile\n```\n\n**Every secret a function declares must hold a value**, or that function refuses to run\n(`409 secret_missing`, naming the ref). Before the push, delete from `collect`\'s `secrets` the sources you do\nnot poll (and their `egressAllow` hosts) \u2014 or store a placeholder value for a source you will add later.\n\nAlso replace the placeholders in `vxil.config.ts`: the OIDC `issuer`, `clientId` and\n`allowedDomains`, the `allowedRedirectOrigins` of your frontend, and `DASHBOARD_URL` in\n`functions/evaluate-alerts.ts`.\n\n| Secret | What it is | Least privilege |\n|---|---|---|\n| `oidc_client_secret` | your identity provider\'s client secret for this app | \u2014 |\n| `stripe_key` | a payments-provider **restricted** key | read on Balance and Charges, nothing else |\n| `shop_token` | the bearer token of the REST API a `rest-json` integration polls | a read-only token |\n| `vxil_peer_key` | an API key of **another** vxil project you own | `usage:read`, plus `cms:read` if you aggregate its collections |\n| `vxil_read_key` | an API key of **this** project | `webhooks:read` only \u2014 received events are not on a function\'s scoped callback |\n| `slack_webhook_url` | a chat incoming-webhook URL | \u2014 |\n\n`vxil_peer_key` and `vxil_read_key` are minted with `vxil keys mint --name <name> --scopes <a,b>` in the project\nthey belong to. `vxil keys mint` acts on your account, so it needs a dashboard session: run `vxil login` once\nfirst (a project key from `vxil quickstart` / `vxil link` is not enough). `vxil link --key` stores the key you give it in `~/.vxil/credentials.json` (so does `vxil login` with its session): treat that file as a secret \u2014 never commit it, copy it into an image or share it.\n\n## 3. Staff-only sign-in, and the roles\n\n`auth` here is **OIDC-only**: `methods.emailPassword` and `methods.magicLink` are off, and\n`providers.oidc.allowedDomains` lists your company domain. A sign-in whose verified e-mail is not on that list\nis refused (`403 oidc_domain_not_allowed`), and with no other method there is no other way in. **That is the\nstaff gate.** Do not turn magic link or e-mail sign-up back on in a project holding company data: these\ncollections have no owner field, every signed-in person reads all of them, so an open sign-up would let any\naddress on the internet read every number on the dashboard.\n\nRoles live in one org. Create it (slug `ops` \u2014 the functions look it up by slug) and the four custom roles once,\nwith a server key that holds `orgs:write`:\n\n```bash\nvxil api POST /v1/orgs --data \'{"slug":"ops","name":"Ops","owner_user_id":"<your user id>"}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"ops-viewer","name":"Viewer","permissions":["dashboard.read"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"operator","name":"Operator","permissions":["dashboard.read","incidents.manage","integrations.sync"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"finance","name":"Finance","permissions":["dashboard.read","revenue.read"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"ops-admin","name":"Ops admin","permissions":["dashboard.read","revenue.read","incidents.manage","integrations.sync","rules.manage","integrations.manage"]}\'\nvxil api POST /v1/orgs/org_\u2026/members --data \'{"user_id":"<user id>","role":"operator"}\'\n```\n\n`viewer` and `admin` are built-in role names with fixed permission sets, which is why the custom ones are\n`ops-viewer` and `ops-admin`. The built-in `owner` holds every permission. The permission strings are yours \u2014\nvxil only answers whether a member holds one (`GET /v1/orgs/{org}/check`).\n\nWith `orgClaims` on, the member\'s role rides the session, and the IdP\'s `groups` claim (named in\n`claims.roles`) rides beside it: a group called `finance` in your identity provider satisfies the same gate as\nthe org role.\n\n## 4. The browser keys\n\nMint two keys in the dashboard under **Keys \u2192 New key** (or with `vxil keys mint` after `vxil login`):\n\n| Key | Class | Scopes | Why |\n|---|---|---|---|\n| `web-auth` | Server | `auth:signin` | signing in is how a session is obtained, so this key cannot require one |\n| `web-data` | Public / thin-client | `cms:read`, `realtime:write`, `functions:invoke` | refused by the edge without a valid staff session |\n\n`web-data` holds **no `cms:write`**. A staff member can read every collection, mint a realtime connect token\nfor a channel (that is what `realtime:write` is for on a thin-client key \u2014 publishing over REST is refused to\nan end-user session) and call functions. Every write \u2014 acknowledging an incident, editing a rule, pausing an\nintegration \u2014 is `POST /v1/fn/ops-action`, which asks orgs whether this person holds the permission before it\nwrites. The write is audited as the project (the function writes with the project\'s cms scope, so the audit\nrow does not name the person); the human is recorded in the row itself \u2014 `acked_by` on an incident:\n\n```ts\nawait vx.asEndUser(session).fn[\'ops-action\']({ op: \'ack\', incident_id: \'itm_\u2026\', note: \'looking\' });\n// 403 { error: \'role_required\', permission: \'incidents.manage\' } for a viewer\n```\n\n| `op` | Permission | Body |\n|---|---|---|\n| `ack`, `resolve` | `incidents.manage` | `{ incident_id, note? }` \u2014 a compare-and-set on the state the button showed |\n| `upsert_rule` | `rules.manage` | `{ rule: { key, series, op, threshold, for_minutes?, severity?, repeat_after_hours?, channels?, title? } }` |\n| `toggle_rule` | `rules.manage` | `{ key, enabled }` |\n| `update_integration` | `integrations.manage` | `{ integration: { key, label?, owner?, poll_every_min?, config? } }` |\n| `pause_integration` | `integrations.manage` | `{ key, paused }` |\n\n"Sync now" is `POST /v1/fn/collect { integration: \'<key>\' }`, gated on `integrations.sync`.\n\n`update_integration` strips every config key that looks like a credential (`key`, `token`, `secret`,\n`password`, `authorization`, \u2026) and every value that looks like a bearer token, and names what it dropped. A\ncredential is a project secret, set with `vxil secrets set`, never from a browser.\n\nThe three settings that decide **where a credential is sent** \u2014 `config.base_url`, `config.auth` and\n`config.peer` \u2014 are not editable through `ops-action` at all: a change answers `409 locked_config` and an edit\nthat leaves them out keeps the stored values. They change only with a server key (`vxil api PATCH\n/v1/cms/items/integrations/<id>`, or a re-seed). `collect` enforces the same thing on its side, in code: each\nsecret is bound to one host (`SECRET_HOST` in `functions/collect.ts` \u2014 `shop_token` \u2192 your API\'s host;\n`stripe_key` \u2192 the payments provider\'s API; peer keys \u2192 the vxil API only, and only the names in\n`PEER_SECRETS`). A row that would send a credential anywhere else turns `broken` without a request being made.\n\n## 5. Adding a source\n\nA source is three edits: a **case in the `adapter` switch** in `functions/collect.ts`, the **host in\n`egressAllow`**, and a **secret**. Then `vxil push`. The egress guard is deny-by-default: a host you forgot\nanswers `403` and the integration turns `broken` with `egress blocked: add <host> to egressAllow and push`.\n\nBefore writing code, try the generic `rest-json` kind \u2014 a row, no code:\n\n```json\n{ "key": "shop", "kind": "rest-json", "enabled": true, "poll_every_min": 5,\n "next_due_at": "2026-01-01T00:00:00Z",\n "config": { "base_url": "https://shop.example.com", "path": "/api/stats",\n "series": [ { "name": "orders_1h", "json_path": "orders.last_hour", "unit": "count" },\n { "name": "revenue_1h", "json_path": "revenue.last_hour", "unit": "usd", "sensitive": true } ] } }\n```\n\nIt sends `Authorization: Bearer <shop_token>` (`"auth": "none"` for a public endpoint) and reads each\n`json_path` (`a.b.0.c`; a trailing `.length` counts an array). Series are named `<integration key>:<name>`.\nThe token goes **only** to the host in `SECRET_HOST.shop_token` at the top of `functions/collect.ts` \u2014 set it to\nyour API\'s host before the push; a `rest-json` row with a token pointing at any other host is refused. A second\nauthenticated REST API is a second secret and its own `case` (below), never a second row reusing `shop_token`.\n\n**A REST API with its own token** \u2014 one secret per vendor, one case:\n\n```ts\ncase \'helpdesk\': {\n const r = await getJson(\'https://api.helpdesk.example/v2/tickets/count?status=open\',\n { authorization: `Bearer ${secrets.helpdesk_token}` }) as { count: number };\n return [{ series: \'helpdesk:open_tickets\', value: r.count, unit: \'count\' }];\n}\n```\n\n**An analytics API that wants a service-account JWT** \u2014 sign it with WebCrypto inside the function (no SDK, no\nNode), then trade it for an access token:\n\n```ts\nasync function serviceToken(sa: { client_email: string; private_key: string }, scope: string): Promise<string> {\n const b64 = (s: string | ArrayBuffer) => btoa(typeof s === \'string\' ? s : String.fromCharCode(...new Uint8Array(s)))\n .replace(/=+$/, \'\').replace(/\\+/g, \'-\').replace(/\\//g, \'_\');\n const now = Math.floor(Date.now() / 1000);\n const unsigned = `${b64(JSON.stringify({ alg: \'RS256\', typ: \'JWT\' }))}.${b64(JSON.stringify({\n iss: sa.client_email, scope, aud: \'https://oauth2.example.com/token\', iat: now, exp: now + 3600 }))}`;\n const der = Uint8Array.from(atob(sa.private_key.replace(/-----[^-]+-----|\\s/g, \'\')), (c) => c.charCodeAt(0));\n const key = await crypto.subtle.importKey(\'pkcs8\', der, { name: \'RSASSA-PKCS1-v1_5\', hash: \'SHA-256\' }, false, [\'sign\']);\n const sig = await crypto.subtle.sign(\'RSASSA-PKCS1-v1_5\', key, new TextEncoder().encode(unsigned));\n const res = await fetch(\'https://oauth2.example.com/token\', { method: \'POST\',\n headers: { \'content-type\': \'application/x-www-form-urlencoded\' },\n body: `grant_type=urn:ietf:params:oauth:grant-type:jwt-bearer&assertion=${unsigned}.${b64(sig)}` });\n return ((await res.json()) as { access_token: string }).access_token;\n}\n```\n\nStore the service-account JSON as one secret, and add both the token host and the API host to `egressAllow`.\n\n**A SQL query over an HTTPS SQL endpoint** \u2014 the function is the SQL client; vxil never connects to your\ndatabase. The pattern, the credential handling and the least-privilege database user are in\n`examples/byo-db` (https://vxil.com/docs/guide/08-running-your-code-functions#your-own-database-from-a-function):\n\n```ts\ncase \'warehouse\': {\n const rows = await sql(secrets.database_url, \'select count(*)::int as n from orders where created_at > now() - interval \\\'1 hour\\\'\');\n return [{ series: \'warehouse:orders_1h\', value: rows[0].n, unit: \'count\' }];\n}\n```\n\n**A tool that pushes instead of being polled** \u2014 create an inbound webhook source and point the tool at its\nreceiver URL; then add an integration row of kind `webhook-source` with `config.source_id`:\n\n```ts\nconst { source_id, receiver_url_path } = await vx.webhooks.sources.create({ provider: \'generic\', name: \'support tool\' });\n// the receiver URL is returned once \u2014 paste it into the tool\'s webhook settings\n```\n\n## 6. Time series on the cms\n\n- **Write closed hours, not samples.** Every collect tick patches the series\' `metric_latest` row (value,\n `updated_at`, the hour\'s sample count and min/max). When a tick finds the stored sample belongs to an hour that\n has ended, it first writes that hour\'s closing sample as one `metric_points` row (`grain: \'hour\'`), then\n records the hour in `last_bucket`.\n- **The bucket is derived.** The `point_bucket` hook derives `bucket` from `ts` (`2026-10-04T13` for an hour,\n `2026-10-04` for a day) and `point_key` concatenates series, grain and bucket into a `unique` field \u2014 so a\n redelivered tick or two overlapping ticks produce a `409`, never a second point. Both are Lane-A `derive`\n hooks, not `validate`: the server computes the value on every write, so a writer that sends a wrong bucket (or\n none) is overwritten rather than refused, and no writer \u2014 `collect`, `compact-daily`, a backfill script \u2014 can\n file a point under the wrong hour. Keep them `derive` in any copy of this config.\n- **Compaction.** `compact-daily` aggregates yesterday\'s hour rows per series on the server (avg, min, max,\n count) into one `grain: \'day\'` row, then deletes hour rows older than 90 days and feed rows older than 30\n (the bounded filtered delete, 100 rows a call). Charts read `grain: \'hour\'` for the last days and\n `grain: \'day\'` beyond.\n- **The 50,000-row scan cap.** A server-side aggregate refuses (`422`) rather than scanning more than 50,000 rows.\n One day of hour rows is `series \xD7 24`, so compaction stays far below it even at 500 series; an aggregate over\n a year of raw 5-minute samples (105,120 per series) would not fit even for one series.\n- **Why not raw points.** Every cms write is an audited row and a change-data frame, and every call a function\n makes back into vxil counts as a request. A sample every 5 minutes is 288 rows a day per series; a closed hour\n is 24 \u2014 twelve times fewer writes, frames and requests, and twelve times fewer rows under every scan.\n\nThe config also shows (commented) the declarative twin of `compact-daily`, a `readModels` aggregate. It is left\noff because the function also compacts the gated value, writes into the collection the charts already read, and\nprunes.\n\n## 7. Sensitive numbers\n\nA series marked `sensitive` (revenue, balances) writes its number to `restricted_value`, which declares\n`readRoles: [\'finance\', \'ops-admin\', \'owner\']`; `value` stays null. A signed-in staff member without one of those\nroles gets the row with `restricted_value` **absent** and cannot filter, sort or group by it. Your server key\nand the functions (which run as the server) still see it, so alerts evaluate it \u2014 but a sensitive series\' number\nnever goes into a chat message, an e-mail or an incident row.\n\n**Change-data frames do not apply `readRoles`.** A `payload: \'full\'` frame carries the whole row to every\nsubscriber of the channel. So `metric_latest` and `incidents` publish **ids only** and the client re-reads the\nrow through the cms, where the gate applies; only `events_feed`, which has no gated field, publishes full rows.\nThe CI gate `cdc-gated-fields` refuses a full frame on any collection that declares a `readRoles` field.\n\n## 8. Alerts once per crossing\n\nA rule is `{ series, op, threshold, for_minutes, severity, repeat_after_hours, channels }`; `op` is `>`, `<`,\n`>=`, `<=` or `stale` (no fresh value for `threshold` minutes). `evaluate-alerts`:\n\n1. on a new breach stores `breach_since` and waits `for_minutes`;\n2. when the breach is sustained, creates the incident under `lock: \'incident:<rule>\'` with\n `guard: { filter: { rule, state: { $in: [\'open\', \'acked\'] } }, max: 1 }` \u2014 of two overlapping ticks exactly\n one gets a `201`, and only that one posts the chat message and sends the e-mail (the built-in `transactional`\n template, to every `ops` member holding `operator`, `ops-admin` or `owner`);\n3. stays quiet while it stays red, except one reminder every `repeat_after_hours`;\n4. if the incident create fails for any other reason (a `5xx`, a hook\'s `422`), leaves the rule `ok` with its\n `breach_since` and counts it in `failed` \u2014 the next tick crosses again, so the alert is delayed, never lost;\n5. on recovery flips the rule back (an `If-Match` on its version picks the one tick that does it), resolves the\n open incident and posts one RESOLVED message.\n\nThe chat poster sends Slack\'s `text` and Discord\'s `content`; for Discord add `discord.com` to `egressAllow`; a\nGoogle Chat space webhook (`chat.googleapis.com`, also added to `egressAllow`) gets `{ text }` alone.\n\n## 9. Reading your other vxil projects\n\nAn integration of kind `vxil-project` reads another project with **that project\'s own key** \u2014 one key per\nproject, minted there with the least it needs (`usage:read` for its request meter, `cms:read` to aggregate its\ncollections). It is pull, not push: the other project does not know this dashboard exists, and revoking the key\nthere disconnects it. A second project gets a second secret: declare `secret:<name>` on `collect`, add the name\nto `PEER_SECRETS` in `functions/collect.ts`, and point the row at it with `config.peer` (any name not on that\nlist is refused, so a row cannot pick some other secret and send it as a bearer key).\n\n```json\n{ "key": "billing-project", "kind": "vxil-project", "enabled": true, "poll_every_min": 60,\n "config": { "aggregates": [ { "name": "signups_24h", "collection": "accounts", "fn": "count", "window_hours": 24 } ] } }\n```\n\n## 10. Limits and plans\n\n| | Free (staging / development projects) | Developer |\n|---|---|---|\n| Functions / crons | 5 functions, 3 crons, fastest every 15 min | 50 functions, 20 crons, every minute |\n| As shipped | set the two `*/5` schedules (`collect`, `evaluate-alerts`) to `*/15`; drop `drain-inbound`\'s cron trigger (its `payments.` trigger stays) | runs as shipped |\n| Requests a month | 100,000 | 1,000,000 |\n\nEvery call a function makes back into vxil counts as a request (vendor calls do not). The arithmetic for 3\npolled integrations \xD7 5 series, a 30-day month:\n\n| Function | Per run | Runs a month | Requests |\n|---|---|---|---|\n| `collect` every 5 min | 1 list + per integration (1 list + 5 tile patches + 1 status patch) = 22 | 8,640 | ~190,000 |\n| hour closes | 15 series \xD7 1 point | 720 hours | ~11,000 |\n| `evaluate-alerts` every 5 min | 1 rules read + 1 latest read (+ writes on crossings only) | 8,640 | ~17,000 |\n| `drain-inbound` every minute, 1 source | 1 list + 1 events read (+ 1 write per event) | 43,200 | ~86,000 + events |\n| `compact-daily` | 2 aggregates + 15 day rows + retention deletes | 30 | ~1,500 |\n\nAbout **305,000 requests a month** on Developer. On Free, at 15-minute cadence with 2 integrations \xD7 4 series and\nno inbound drain: `collect` ~37,000, `evaluate-alerts` ~6,000, hour closes ~6,000 \u2014 about **50,000**. A tile list\nread by your frontend counts too, so let realtime change-data tell the page when to re-read instead of polling.\n\n## 11. Compose with\n\n- **alerts-to-slack** \u2014 alerts on the platform\'s own failure events (a dead-lettered job, a quarantined\n function, a failing payments webhook) next to this dashboard\'s business thresholds.\n- **service-monitor** \u2014 uptime checks for your endpoints, written into the same kind of rows.\n- **team-workspace** \u2014 the full org model: invitations, per-resource grants, a field only finance receives.\n\n## 12. Where the frontend lives\n\nThe **App builder** builds public-facing apps. An internal dashboard is your own frontend \u2014 any framework,\nany host \u2014 reading this project with the two keys above. The reference is `examples/apps/team-dashboard`:\nsign-in through your identity provider, tiles that re-read on change-data frames, trend charts over\n`metric_points`, the feed, and the incident buttons calling `ops-action`.\n',
18473
+ "functions": {
18474
+ "collect.ts": "// collect.ts \u2014 PULL: poll the team's services into the tiles and the hourly trend\n// (a vxil function; cron every 5 minutes + an http \"Sync now\" door).\n//\n// There is no connectors feature. A connector is the `adapter` switch below +\n// a secret + a host in egressAllow. Each tick:\n// 1. reads the enabled integrations whose next_due_at has passed (oldest\n// first, at most MAX_PER_TICK);\n// 2. runs that integration's adapter \u2014 stripe, rest-json or vxil-project \u2014\n// which returns samples: { series, value, unit, sensitive };\n// 3. per series: when the hour of the stored sample has CLOSED, writes that\n// hour's closing sample as ONE metric_points row (grain 'hour'; the unique\n// point_key makes a redelivered tick a 409, never a duplicate) and records\n// it in last_bucket; then PATCHes metric_latest with the new sample (the\n// first sighting creates it);\n// 4. PATCHes the integration: ok + next_due_at, or broken + the error.\n//\n// A sensitive series (revenue, balances) writes `restricted_value` \u2014 readable\n// only by the roles in the config's readRoles \u2014 and leaves `value` null.\n//\n// A vendor failure never throws: the integration is marked broken (failures\n// counted, last_error kept to 500 chars) and the tick answers 200, so one\n// vendor's outage does not trip the function's circuit breaker.\n//\n// \"Sync now\" (http): `POST /v1/fn/collect { integration: '<key>' }` with a\n// signed-in staff session. The caller must hold `integrations.sync` in the ops\n// org (checked with orgs, not trusted from the browser). The http run updates\n// the tiles; a series whose hour has closed is left to the next cron tick,\n// which can read the restricted value it needs to close that hour.\n// A server caller (no session \u2014 `vxil functions invoke collect`, your backend)\n// runs one named integration, or the whole tick when none is named.\n//\n// cron-walk: drains-filter \u2014 each processed integration is PATCHed with next_due_at = now + poll_every_min, so it leaves the read filter\n\nimport type { CronFunctionEnvelope, HttpFunctionEnvelope } from '@vxil/sdk';\n\nconst MAX_PER_TICK = 20; // integrations per tick\nconst MAX_SERIES_PER_SOURCE = 25; // samples one adapter may emit\nconst ERROR_MAX = 500;\nconst OPS_ORG_SLUG = 'ops'; // the one org every staff member joins (README step 3)\nconst VENDOR_TIMEOUT_MS = 8_000;\n\n// \u2500\u2500 credentials are bound to hosts HERE, in code \u2014 never by a row \u2500\u2500\n// An integration row is data a dashboard admin can edit, and its base_url may\n// name any host in egressAllow. So a row never decides where a credential goes:\n// each secret is sent to the ONE host written below, and a row that points it\n// anywhere else turns `broken` without a request being made.\n// shop_token \u2192 the host of your REST API (edit to yours; the demo row uses \"auth\": \"none\")\n// stripe_key \u2192 api.stripe.com, hardcoded in its adapter\n// peer keys \u2192 this platform's own API only, and only the names listed in PEER_SECRETS\nconst SECRET_HOST: Record<string, string> = { shop_token: 'shop.example.com' };\n// The secrets a `vxil-project` row may name in config.peer. A second project =\n// a second `secret:<name>` on collect AND its name here.\nconst PEER_SECRETS: readonly string[] = ['vxil_peer_key'];\n\ntype Envelope = CronFunctionEnvelope | HttpFunctionEnvelope<{ integration?: string }>;\n\ninterface SeriesDef { name: string; json_path: string; unit?: string; sensitive?: boolean; label?: string }\ninterface AggregateDef { name: string; collection: string; fn: 'count' | 'sum' | 'avg' | 'min' | 'max'; field?: string; filter?: Record<string, unknown>; window_hours?: number; sensitive?: boolean; unit?: string }\ninterface IntegrationConfig {\n base_url?: string; path?: string; auth?: 'bearer' | 'none'; series?: SeriesDef[];\n aggregates?: AggregateDef[]; peer?: string;\n}\ninterface Integration {\n key: string; kind: string; status?: string; enabled?: boolean; poll_every_min?: number;\n consecutive_failures?: number; config?: IntegrationConfig; label?: string;\n}\ninterface Latest {\n series: string; source?: string; status?: string; last_bucket?: string | null; value?: number | null;\n restricted_value?: number | null; updated_at?: string; label?: string; unit?: string; sensitive?: boolean;\n hour_samples?: number; hour_min?: number | null; hour_max?: number | null;\n}\ninterface Item<T> { item_id: string; version?: number; data: T }\ninterface Sample { series: string; value: number; unit?: string | undefined; label?: string | undefined; sensitive?: boolean | undefined }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Envelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cmsJwt = env.scoped_jwts?.cms;\n if (!cmsJwt) return json({ error: 'missing cms scope' }, 403);\n const cms = new Cms(base, cmsJwt);\n const now = new Date();\n\n // \u2500\u2500 the http door \u2500\u2500\n // a signed-in staff member (\"Sync now\"): checked against the ops org, one integration\n // a server caller (`vxil functions invoke collect`, your own backend): one\n // integration when named, otherwise exactly what a cron tick does\n const key = env.trigger === 'http' && typeof env.payload?.integration === 'string' ? env.payload.integration : '';\n if (env.trigger === 'http' && (env.end_user || key)) {\n const user = env.end_user;\n if (user && !(await orgAllows(base, env.scoped_jwts?.orgs, user.id, 'integrations.sync'))) {\n return json({ error: 'role_required', permission: 'integrations.sync' }, 403);\n }\n const row = key ? (await cms.list<Integration>('integrations', { key }, { limit: 1 })).items[0] : undefined;\n if (!row) return json({ error: 'unknown_integration' }, 404);\n if (row.data.kind === 'webhook-source') return json({ error: 'webhook sources are drained, not polled' }, 422);\n const out = await pollOne(cms, env.secrets ?? {}, base, row, now, user ? 'http' : 'cron');\n return json(out, 200);\n }\n\n // \u2500\u2500 the cron tick \u2500\u2500\n const due = await cms.list<Integration>('integrations', {\n enabled: true,\n kind: { $in: ['stripe', 'rest-json', 'vxil-project'] },\n next_due_at: { $lte: now.toISOString() },\n }, { sort: 'next_due_at', limit: MAX_PER_TICK });\n const results = [];\n for (const row of due.items) results.push(await pollOne(cms, env.secrets ?? {}, base, row, now, 'cron'));\n return json({\n polled: results.length,\n ok: results.filter((r) => r.ok).length,\n broken: results.filter((r) => !r.ok).map((r) => r.integration),\n series: results.reduce((n, r) => n + r.series, 0),\n points: results.reduce((n, r) => n + r.points, 0),\n // more were due than one tick takes; the rest are first in line next tick\n truncated: due.items.length === MAX_PER_TICK,\n }, 200);\n },\n};\n\n/** One integration: adapter \u2192 samples \u2192 tiles + closed hours \u2192 the integration's own row. */\nasync function pollOne(\n cms: Cms, secrets: Record<string, string>, base: string, row: Item<Integration>, now: Date, mode: 'cron' | 'http',\n): Promise<{ integration: string; ok: boolean; series: number; points: number; deferred: number; error?: string }> {\n const integ = row.data;\n const nextDue = new Date(now.getTime() + Math.max(1, integ.poll_every_min ?? 5) * 60_000).toISOString();\n let samples: Sample[];\n try {\n samples = (await adapter(integ, secrets, base, now)).slice(0, MAX_SERIES_PER_SOURCE);\n } catch (e) {\n const missing = e instanceof MissingSecret;\n const error = String((e as Error).message ?? e).slice(0, ERROR_MAX);\n // If-Match: the row was read this tick, so the failure count stays exact\n // when two ticks overlap (the loser's PATCH is a 409 and changes nothing).\n await cms.patch('integrations', row.item_id, {\n status: missing ? 'could_not_check' : 'broken',\n consecutive_failures: (integ.consecutive_failures ?? 0) + 1,\n last_error: error,\n last_error_at: now.toISOString(),\n next_due_at: nextDue,\n }, row.version);\n return { integration: integ.key, ok: false, series: 0, points: 0, deferred: 0, error };\n }\n\n const w = await writeSamples(cms, integ.key, samples, now, mode);\n await cms.patch('integrations', row.item_id, {\n status: 'ok', consecutive_failures: 0, last_ok_at: now.toISOString(), last_error: null, next_due_at: nextDue,\n }, row.version);\n return { integration: integ.key, ok: true, series: w.series, points: w.points, deferred: w.deferred };\n}\n\n/** Write one source's samples: close finished hours, then update the tiles. */\nasync function writeSamples(\n cms: Cms, source: string, samples: Sample[], now: Date, mode: 'cron' | 'http',\n): Promise<{ series: number; points: number; deferred: number }> {\n const nowIso = now.toISOString();\n const hourNow = nowIso.slice(0, 13);\n const existing = new Map<string, Item<Latest>>();\n for (const it of (await cms.list<Latest>('metric_latest', { source }, { limit: 100 })).items) existing.set(it.data.series, it);\n\n let series = 0;\n let points = 0;\n let deferred = 0;\n for (const s of samples) {\n const cur = existing.get(s.series);\n const sensitive = Boolean(s.sensitive);\n const reading = sensitive ? { value: null, restricted_value: s.value } : { value: s.value, restricted_value: null };\n\n if (!cur) {\n // first sighting \u2014 a racing tick's create is a 409; the next tick patches it\n const r = await cms.create('metric_latest', {\n series: s.series, source, status: 'ok', label: s.label ?? s.series, unit: s.unit ?? null, sensitive,\n ...(sensitive ? { restricted_value: s.value } : { value: s.value, hour_min: s.value, hour_max: s.value }),\n hour_samples: 1, updated_at: nowIso,\n });\n if (r.ok) series++;\n continue;\n }\n\n const prevHour = (cur.data.updated_at ?? '').slice(0, 13);\n const hourClosed = prevHour !== '' && prevHour < hourNow;\n const patch: Record<string, unknown> = {\n ...reading, updated_at: nowIso, label: s.label ?? cur.data.label ?? s.series, unit: s.unit ?? cur.data.unit ?? null, sensitive,\n };\n if (cur.data.status === 'stale') patch.status = 'ok'; // fresh data again; alert severities are evaluate-alerts' to set\n\n if (hourClosed) {\n if (mode === 'http') { deferred++; continue; } // the cron closes this hour first\n if (cur.data.last_bucket !== prevHour) {\n // the closed hour's closing sample \u2192 ONE point (409 on point_key = already written)\n const wasSensitive = Boolean(cur.data.sensitive);\n const closing = wasSensitive ? cur.data.restricted_value : cur.data.value;\n if (typeof closing === 'number') {\n const p = await cms.create('metric_points', {\n series: s.series, grain: 'hour', ts: `${prevHour}:00:00.000Z`,\n // bucket + point_key are derived by the config's hooks; sent too so the\n // row is complete even where hooks are switched off\n bucket: prevHour, point_key: `${s.series}|hour|${prevHour}`,\n ...(wasSensitive\n ? { restricted_value: closing }\n : { value: closing, min: cur.data.hour_min ?? closing, max: cur.data.hour_max ?? closing }),\n samples: cur.data.hour_samples ?? 1, unit: cur.data.unit ?? null,\n });\n if (p.ok) points++;\n if (p.ok || p.status === 409) patch.last_bucket = prevHour;\n } else {\n patch.last_bucket = prevHour; // nothing to close (the stored value was gated or empty)\n }\n }\n patch.hour_samples = 1;\n patch.hour_min = sensitive ? null : s.value;\n patch.hour_max = sensitive ? null : s.value;\n } else {\n patch.hour_samples = (cur.data.hour_samples ?? 0) + 1;\n patch.hour_min = sensitive ? null : Math.min(cur.data.hour_min ?? s.value, s.value);\n patch.hour_max = sensitive ? null : Math.max(cur.data.hour_max ?? s.value, s.value);\n }\n const r = await cms.patch('metric_latest', cur.item_id, patch, cur.version);\n if (r.ok) series++;\n }\n return { series, points, deferred };\n}\n\n// \u2500\u2500 the adapters: one case per kind, self-contained (a blueprint is a clone) \u2500\u2500\n\nclass MissingSecret extends Error {}\n\nasync function adapter(integ: Integration, secrets: Record<string, string>, vxilBase: string, now: Date): Promise<Sample[]> {\n const cfg = integ.config ?? {};\n switch (integ.kind) {\n case 'stripe': {\n // A RESTRICTED key, read-only on Balance and Charges. Amounts are minor units.\n const key = secrets.stripe_key;\n if (!key) throw new MissingSecret('secret stripe_key is not set');\n const auth = { authorization: `Bearer ${key}` };\n const balance = await getJson(`https://api.stripe.com/v1/balance`, auth) as { available?: Array<{ amount: number; currency: string }> };\n const since = Math.floor(now.getTime() / 1000) - 3600;\n const charges = await getJson(`https://api.stripe.com/v1/charges?limit=100&created[gte]=${since}`, auth) as {\n data?: Array<{ amount: number; currency: string; status: string }>; has_more?: boolean;\n };\n const list = charges.data ?? [];\n const ok = list.filter((c) => c.status === 'succeeded');\n const failed = list.filter((c) => c.status === 'failed').length;\n const out: Sample[] = [\n { series: 'stripe:charges_1h', value: ok.length, unit: 'count', label: 'Charges (last hour)' },\n { series: 'stripe:failed_charges_1h', value: failed, unit: 'count', label: 'Failed charges (last hour)' },\n {\n series: 'stripe:charge_error_rate_1h', unit: '%', label: 'Charge error rate (last hour)',\n value: list.length === 0 ? 0 : Math.round((failed / list.length) * 1000) / 10,\n },\n ];\n const volume = new Map<string, number>();\n for (const c of ok) volume.set(c.currency, (volume.get(c.currency) ?? 0) + c.amount);\n for (const [cur, amount] of volume) {\n out.push({ series: `stripe:volume_1h:${cur}`, value: amount / 100, unit: cur, label: `Volume ${cur.toUpperCase()} (last hour)`, sensitive: true });\n }\n for (const b of balance.available ?? []) {\n out.push({ series: `stripe:balance:${b.currency}`, value: b.amount / 100, unit: b.currency, label: `Available ${b.currency.toUpperCase()}`, sensitive: true });\n }\n // more than 100 charges in the hour: the counts are a floor \u2014 page with\n // `starting_after` (and keep the cursor on the integration row) if you need exact ones\n return out;\n }\n\n case 'rest-json': {\n // Any JSON endpoint: GET base_url + path, pick numbers with json_path.\n if (!cfg.base_url || !cfg.series?.length) throw new Error('config needs base_url and series[]');\n let url: URL;\n try { url = new URL(`${cfg.base_url.replace(/\\/$/, '')}${cfg.path ?? ''}`); } catch { throw new Error('config base_url + path is not a URL'); }\n if (url.protocol !== 'https:') throw new Error('config base_url must be https://');\n const headers: Record<string, string> = { accept: 'application/json' };\n if (cfg.auth !== 'none') {\n // the host of the FINAL url (a path like \"@other.host\" cannot move it)\n if (url.host !== SECRET_HOST.shop_token) {\n throw new Error(`shop_token is bound to ${SECRET_HOST.shop_token} and is never sent to ${url.host}: set \"auth\": \"none\" for a public endpoint, or change SECRET_HOST in collect.ts`);\n }\n if (!secrets.shop_token) throw new MissingSecret('secret shop_token is not set');\n headers.authorization = `Bearer ${secrets.shop_token}`;\n }\n const body = await getJson(url.href, headers);\n const out: Sample[] = [];\n for (const s of cfg.series) {\n const v = pick(body, s.json_path);\n if (typeof v === 'number' && Number.isFinite(v)) {\n out.push({ series: `${integ.key}:${s.name}`, value: v, unit: s.unit, label: s.label ?? s.name, sensitive: s.sensitive });\n }\n }\n if (out.length === 0) throw new Error(`no numeric value at ${cfg.series.map((s) => s.json_path).join(', ')}`);\n return out;\n }\n\n case 'vxil-project': {\n // Another vxil project of yours, read with ITS key (one secret per project;\n // the least-privilege pattern \u2014 usage:read, plus cms:read for aggregates).\n const secretName = cfg.peer ?? 'vxil_peer_key';\n if (!PEER_SECRETS.includes(secretName)) {\n throw new Error(`config.peer '${secretName}' is not a peer secret: add it to PEER_SECRETS in collect.ts`);\n }\n const key = secrets[secretName];\n if (!key) throw new MissingSecret(`secret ${secretName} is not set`);\n const H = { authorization: `Bearer ${key}`, 'content-type': 'application/json' };\n const usage = await getJson(`${vxilBase}/v1/usage`, H) as { data?: { requests?: { used?: number } } };\n const out: Sample[] = [];\n if (typeof usage.data?.requests?.used === 'number') {\n out.push({ series: `${integ.key}:requests_mtd`, value: usage.data.requests.used, unit: 'requests', label: 'Requests this month' });\n }\n for (const a of cfg.aggregates ?? []) {\n const since = new Date(now.getTime() - (a.window_hours ?? 24) * 3600_000).toISOString();\n const r = await fetch(`${vxilBase}/v1/cms/items/${encodeURIComponent(a.collection)}/aggregate`, {\n method: 'POST', headers: H, signal: AbortSignal.timeout(VENDOR_TIMEOUT_MS),\n body: JSON.stringify({\n aggregates: [a.fn === 'count' ? { fn: 'count', as: 'v' } : { fn: a.fn, field: a.field, as: 'v' }],\n ...(a.filter ? { filter: a.filter } : {}),\n window: { field: 'created_at', since },\n }),\n });\n if (!r.ok) throw new Error(`aggregate ${a.collection}: ${r.status}`);\n const g = ((await r.json()) as { data?: { groups?: Array<{ v?: number }> } }).data?.groups?.[0];\n out.push({ series: `${integ.key}:${a.name}`, value: Number(g?.v ?? 0), unit: a.unit, label: a.name, sensitive: a.sensitive });\n }\n return out;\n }\n\n default:\n throw new Error(`no adapter for kind '${integ.kind}'`);\n }\n}\n\nasync function getJson(url: string, headers: Record<string, string>): Promise<unknown> {\n const r = await fetch(url, { headers, signal: AbortSignal.timeout(VENDOR_TIMEOUT_MS) });\n if (r.status === 403 && r.headers.get('x-vxil-egress') === 'blocked') {\n throw new Error(`egress blocked: add ${new URL(url).hostname} to egressAllow and push`);\n }\n if (!r.ok) throw new Error(`GET ${new URL(url).hostname}${new URL(url).pathname}: ${r.status}`);\n return r.json();\n}\n\n/** 'a.b.0.c' into a JSON value; a trailing '.length' counts an array. */\nfunction pick(root: unknown, path: string): unknown {\n let cur: unknown = root;\n for (const part of path.split('.')) {\n if (part === 'length' && Array.isArray(cur)) return cur.length;\n if (cur === null || typeof cur !== 'object') return undefined;\n cur = (cur as Record<string, unknown>)[part];\n }\n return cur;\n}\n\n/** Revocation-grade check: is this user allowed `permission` in the ops org? */\nasync function orgAllows(base: string, orgsJwt: string | undefined, userId: string, permission: string): Promise<boolean> {\n if (!orgsJwt) return false;\n const H = { authorization: `Bearer ${orgsJwt}` };\n const mine = await fetch(`${base}/v1/orgs?user_id=${encodeURIComponent(userId)}`, { headers: H });\n if (!mine.ok) return false;\n const org = ((await mine.json()) as { data?: { orgs?: Array<{ org_id: string; slug: string }> } }).data?.orgs\n ?.find((o) => o.slug === OPS_ORG_SLUG);\n if (!org) return false;\n const q = `user_id=${encodeURIComponent(userId)}&permission=${encodeURIComponent(permission)}`;\n const check = await fetch(`${base}/v1/orgs/${encodeURIComponent(org.org_id)}/check?${q}`, { headers: H });\n if (!check.ok) return false;\n return ((await check.json()) as { data?: { allowed?: boolean } }).data?.allowed === true;\n}\n\n/** The cms REST envelope: reads { data: { items: [{ item_id, version, data }], next_cursor } }. */\nclass Cms {\n constructor(private base: string, private jwt: string) {}\n private h(extra: Record<string, string> = {}) {\n return { authorization: `Bearer ${this.jwt}`, 'content-type': 'application/json', ...extra };\n }\n async list<T>(coll: string, filter: Record<string, unknown>, o: { sort?: string; limit?: number } = {}): Promise<{ items: Item<T>[]; next_cursor: string | null }> {\n const q = new URLSearchParams({ filter: JSON.stringify(filter), limit: String(o.limit ?? 100) });\n if (o.sort) q.set('sort', o.sort);\n const r = await fetch(`${this.base}/v1/cms/items/${coll}?${q}`, { headers: this.h() });\n if (!r.ok) throw new Error(`cms list ${coll}: ${r.status}`);\n const d = ((await r.json()) as { data?: { items?: Item<T>[]; next_cursor?: string | null } }).data;\n return { items: d?.items ?? [], next_cursor: d?.next_cursor ?? null };\n }\n async create(coll: string, data: Record<string, unknown>): Promise<{ ok: boolean; status: number }> {\n const r = await fetch(`${this.base}/v1/cms/items/${coll}`, {\n method: 'POST', headers: this.h(), body: JSON.stringify({ status: 'published', data }),\n });\n return { ok: r.ok, status: r.status };\n }\n async patch(coll: string, id: string, data: Record<string, unknown>, version?: number): Promise<{ ok: boolean; status: number }> {\n const r = await fetch(`${this.base}/v1/cms/items/${coll}/${id}`, {\n method: 'PATCH',\n headers: this.h(version !== undefined ? { 'if-match': String(version) } : {}),\n body: JSON.stringify({ data }),\n });\n return { ok: r.ok, status: r.status };\n }\n}\n\nconst json = (o: unknown, status: number) => Response.json(o, { status });\n",
18475
+ "compact-daily.ts": "// compact-daily.ts \u2014 hour rows \u2192 day rows, then retention (a vxil function;\n// cron '15 0 * * *', just after midnight UTC).\n//\n// 1. Two bounded server-side aggregates over YESTERDAY's hour rows\n// (POST /v1/cms/items/metric_points/aggregate, grouped by series, windowed\n// on the `ts` slot): avg/min/max/count of `value`, and avg/max of the gated\n// `restricted_value` (this function runs as the server, so it sees both).\n// 2. One grain 'day' row per series. The unique point_key ('series|day|date')\n// makes a re-run a 409 per row \u2014 the day is never written twice.\n// 3. Retention: hour rows older than RETAIN_HOUR_DAYS and feed rows older than\n// RETAIN_FEED_DAYS are removed with the bounded filtered delete (100 rows a\n// call, following next_cursor, at most MAX_DELETE_CALLS calls a night; the\n// remainder is reported as `truncated` and taken the next night).\n// 4. The `__compact__` row in ops_state records the last day finished.\n//\n// Why hour buckets and not raw samples: every cms write is an audited row and\n// a change-data frame. One closed hour per series is 24 rows a day; a raw\n// sample every 5 minutes would be 288 \u2014 twelve times the writes, the frames,\n// the request budget, and the rows the 50,000-row aggregate scan must cover.\n//\n// cron-walk: complete-per-tick \u2014 one aggregate per value kind is the whole read; the retention deletes follow next_cursor under MAX_DELETE_CALLS and report `truncated`\n\nimport type { CronFunctionEnvelope } from '@vxil/sdk';\n\nconst RETAIN_HOUR_DAYS = 90;\nconst RETAIN_FEED_DAYS = 30;\nconst MAX_DELETE_CALLS = 30;\nconst MAX_GROUPS = 500;\n\ninterface Group { key: { series?: string }; avg?: number | null; min?: number | null; max?: number | null; n?: number }\ninterface Item<T> { item_id: string; version?: number; data: T }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as CronFunctionEnvelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cmsJwt = env.scoped_jwts?.cms;\n if (!cmsJwt) return Response.json({ error: 'missing cms scope' }, { status: 403 });\n const H = { authorization: `Bearer ${cmsJwt}`, 'content-type': 'application/json' };\n\n // the slot this tick was due for (stable across a late start or a re-delivery)\n const slot = new Date(env.scheduled_for ?? Date.now());\n const day = new Date(Date.UTC(slot.getUTCFullYear(), slot.getUTCMonth(), slot.getUTCDate() - 1)).toISOString().slice(0, 10);\n const since = `${day}T00:00:00.000Z`;\n const until = new Date(Date.parse(since) + 86_400_000).toISOString(); // exclusive\n\n const aggregate = async (field: 'value' | 'restricted_value'): Promise<Group[] | null> => {\n const r = await fetch(`${base}/v1/cms/items/metric_points/aggregate`, {\n method: 'POST', headers: H,\n body: JSON.stringify({\n aggregates: [\n { fn: 'avg', field, as: 'avg' }, { fn: 'min', field, as: 'min' },\n { fn: 'max', field, as: 'max' }, { fn: 'count', as: 'n' },\n ],\n groupBy: ['series'],\n filter: { grain: 'hour' },\n window: { field: 'ts', since, until },\n limit: MAX_GROUPS,\n }),\n });\n if (!r.ok) return null;\n return ((await r.json()) as { data?: { groups?: Group[] } }).data?.groups ?? [];\n };\n const plain = await aggregate('value');\n const gated = await aggregate('restricted_value');\n if (!plain || !gated) return Response.json({ error: 'aggregate_failed', day }, { status: 502 });\n\n // one day row per series; a series whose hours carried restricted_value is sensitive\n const bySeries = new Map<string, { plain?: Group; gated?: Group }>();\n for (const g of plain) if (g.key.series) bySeries.set(g.key.series, { plain: g });\n for (const g of gated) if (g.key.series) bySeries.set(g.key.series, { ...bySeries.get(g.key.series), gated: g });\n\n let written = 0;\n let existed = 0;\n for (const [series, { plain: p, gated: q }] of bySeries) {\n const sensitive = typeof q?.avg === 'number';\n if (!sensitive && typeof p?.avg !== 'number') continue;\n const r = await fetch(`${base}/v1/cms/items/metric_points`, {\n method: 'POST', headers: H,\n body: JSON.stringify({\n status: 'published',\n data: {\n series, grain: 'day', ts: since, bucket: day, point_key: `${series}|day|${day}`,\n samples: p?.n ?? q?.n ?? 0,\n ...(sensitive\n ? { restricted_value: round(q!.avg!) }\n : { value: round(p!.avg!), min: p!.min ?? null, max: p!.max ?? null }),\n },\n }),\n });\n if (r.ok) written++;\n else if (r.status === 409) existed++;\n }\n\n // retention \u2014 the bounded filtered delete, followed to the end or the cap\n const cutoff = (days: number) => new Date(Date.parse(since) - days * 86_400_000).toISOString();\n let calls = 0;\n const sweep = async (coll: string, filter: Record<string, unknown>): Promise<{ deleted: number; complete: boolean }> => {\n let deleted = 0;\n let cursor: string | null = null;\n while (calls < MAX_DELETE_CALLS) {\n calls++;\n const r = await fetch(`${base}/v1/cms/items/${coll}/delete`, {\n method: 'POST', headers: H,\n body: JSON.stringify({ filter, limit: 100, ...(cursor ? { cursor } : {}) }),\n });\n if (!r.ok) return { deleted, complete: false };\n const d = ((await r.json()) as { data?: { deleted?: number; complete?: boolean; next_cursor?: string | null } }).data ?? {};\n deleted += d.deleted ?? 0;\n // a call stopped by its time budget hands back next_cursor; otherwise the\n // filter simply re-matches what is left \u2014 loop while rows are still going\n cursor = d.complete === false ? (d.next_cursor ?? null) : null;\n if ((d.deleted ?? 0) === 0) return { deleted, complete: true };\n }\n return { deleted, complete: false };\n };\n const hours = await sweep('metric_points', { grain: 'hour', ts: { $lt: cutoff(RETAIN_HOUR_DAYS) } });\n const feed = await sweep('events_feed', { occurred_at: { $lt: cutoff(RETAIN_FEED_DAYS) } });\n\n // remember the finished day (create once, then PATCH)\n const sq = new URLSearchParams({ filter: JSON.stringify({ key: '__compact__' }), limit: '1' });\n const st = await fetch(`${base}/v1/cms/items/ops_state?${sq}`, { headers: H });\n const row = ((await st.json().catch(() => ({}))) as { data?: { items?: Item<{ cursor?: string }>[] } }).data?.items?.[0];\n const mark = { key: '__compact__', cursor: day, updated_at: new Date().toISOString() };\n if (row) {\n await fetch(`${base}/v1/cms/items/ops_state/${row.item_id}`, { method: 'PATCH', headers: H, body: JSON.stringify({ data: mark }) });\n } else {\n await fetch(`${base}/v1/cms/items/ops_state`, { method: 'POST', headers: H, body: JSON.stringify({ status: 'published', data: mark }) });\n }\n\n return Response.json({\n day, series: bySeries.size, written, existed,\n groups_capped: plain.length >= MAX_GROUPS || gated.length >= MAX_GROUPS,\n pruned: { hour_rows: hours.deleted, feed_rows: feed.deleted },\n truncated: !hours.complete || !feed.complete, // retention resumes tomorrow\n });\n },\n};\n\nconst round = (n: number) => Math.round(n * 1000) / 1000;\n",
18476
+ "drain-inbound.ts": "// drain-inbound.ts \u2014 PUSH: received webhooks and payments events \u2192 the updates\n// feed (a vxil function; cron every minute + a `payments.` event trigger).\n//\n// Two front doors, told apart by the envelope's `trigger`:\n//\n// cron For every enabled `webhook-source` integration (an inbound webhook\n// source you created with `POST /v1/webhooks/sources`), read the\n// events received since the integration's watermark and write one\n// events_feed row each. The received-events list is newest-first\n// with a `cursor` that pages OLDER, so the drain pages back until it\n// reaches the watermark (bounded by MAX_PAGES \u2014 a backlog past that\n// is reported as `gap`), writes oldest-first, then advances the\n// watermark on the integration row with If-Match.\n// The list is not on a function's scoped callback: it is read with\n// `vxil_read_key`, a key of THIS project holding only webhooks:read.\n//\n// webhook The project's own payments-integration events (`payments.*`)\n// arrive here within seconds, one invocation per event.\n//\n// Every row's ext_id is the sender's event id, declared unique: a redelivered\n// event, an overlapping tick or a lost watermark advance is a 409 \u2014 counted as\n// a duplicate, never a second row. Only a title and a short summary are kept:\n// a vendor payload can carry personal data, and this feed's change-data\n// frames are full rows.\n//\n// cron-walk: persisted-cursor \u2014 the newest drained event id is the integration row's `cursor` watermark, advanced with If-Match\n\nimport type { CronFunctionEnvelope, WebhookFunctionEnvelope } from '@vxil/sdk';\n\nconst MAX_SOURCES = 25;\nconst MAX_PAGES = 3; // \xD7 200 events per source per tick\nconst FIRST_RUN_EVENTS = 20; // a new source shows its latest 20, not its whole history\n\ntype Envelope = CronFunctionEnvelope | WebhookFunctionEnvelope;\ninterface Integration { key: string; kind: string; enabled?: boolean; cursor?: string | null; status?: string; config?: { source_id?: string; link_url?: string } }\ninterface Item<T> { item_id: string; version?: number; data: T }\ninterface Received { event_id: string; source_id: string; provider: string; provider_event_id: string | null; event_type: string | null; received_at: string; sig_verified?: boolean }\ninterface FeedRow { source: string; kind: string; severity: 'info' | 'warn' | 'error'; ext_id: string; occurred_at: string; title: string; url?: string | null; summary?: string }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Envelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cmsJwt = env.scoped_jwts?.cms;\n if (!cmsJwt) return Response.json({ error: 'missing cms scope' }, { status: 403 });\n const H = { authorization: `Bearer ${cmsJwt}`, 'content-type': 'application/json' };\n\n // \u2500\u2500 the payments integration's own events, pushed \u2500\u2500\n if (env.trigger === 'webhook') {\n const p = env.payload;\n const d = (p.data && !('truncated' in p.data) ? p.data : {}) as Record<string, unknown>;\n const row: FeedRow = {\n source: 'payments',\n kind: p.event,\n severity: /failed|disputed|past_due|rejected/.test(p.event) ? (/failed|rejected/.test(p.event) ? 'error' : 'warn') : 'info',\n ext_id: `audit:${p.audit_id ?? env.idempotency_key}`,\n occurred_at: p.occurred_at ?? new Date().toISOString(),\n title: p.event.replace(/^payments\\./, 'payments: ').replace(/[._]/g, ' ').slice(0, 200),\n // provider and status only \u2014 never the end user, an amount (the feed is\n // ungated and its frames are full rows; revenue lives in restricted_value)\n // or the raw payload\n summary: [str(d.provider), str(d.status)].filter(Boolean).join(' \xB7 ').slice(0, 2000),\n };\n const res = await createFeed(base, H, row);\n return Response.json({ trigger: 'webhook', event: p.event, written: res === 'created', duplicate: res === 'duplicate' });\n }\n\n // \u2500\u2500 the cron drain \u2500\u2500\n const readKey = env.secrets?.vxil_read_key;\n if (!readKey) return Response.json({ error: 'missing secret vxil_read_key' }, { status: 409 });\n const q = new URLSearchParams({ filter: JSON.stringify({ kind: 'webhook-source', enabled: true }), limit: String(MAX_SOURCES) });\n const lr = await fetch(`${base}/v1/cms/items/integrations?${q}`, { headers: H });\n if (!lr.ok) return Response.json({ error: 'integrations_read_failed', status: lr.status }, { status: 502 });\n const sources = ((await lr.json()) as { data?: { items?: Item<Integration>[] } }).data?.items ?? [];\n\n const out = { sources: sources.length, written: 0, duplicates: 0, advanced: 0, gaps: [] as string[], failed: [] as string[] };\n for (const src of sources) {\n const sourceId = src.data.config?.source_id;\n if (!sourceId) continue;\n const watermark = src.data.cursor ?? null;\n const fresh: Received[] = [];\n let page: string | null = null;\n let reached = false;\n let ok = true;\n for (let i = 0; i < MAX_PAGES; i++) {\n const eq = new URLSearchParams({ source_id: sourceId, limit: watermark ? '200' : String(FIRST_RUN_EVENTS) });\n if (page) eq.set('cursor', page);\n const r = await fetch(`${base}/v1/webhooks/events?${eq}`, { headers: { authorization: `Bearer ${readKey}` } });\n if (!r.ok) { ok = false; break; }\n const d = ((await r.json()) as { data?: { events?: Received[]; next_cursor?: string | null } }).data;\n for (const e of d?.events ?? []) {\n // event ids sort by time; everything at or below the watermark is already in the feed\n if (watermark && e.event_id <= watermark) { reached = true; break; }\n fresh.push(e);\n }\n page = d?.next_cursor ?? null;\n if (reached || !page || !watermark) break; // first run: the newest page only\n }\n if (!ok) {\n out.failed.push(src.data.key);\n if (src.data.status !== 'broken') await patchIntegration(base, H, src, { status: 'broken', last_error: 'received-events read failed', last_error_at: new Date().toISOString() });\n continue;\n }\n if (watermark && !reached && page) out.gaps.push(src.data.key); // older than MAX_PAGES pages: skipped, reported\n\n for (const e of fresh.reverse()) { // oldest first, so the feed reads in order\n const res = await createFeed(base, H, {\n source: src.data.key,\n kind: (e.event_type ?? 'event').slice(0, 100),\n severity: /fail|error|dispute|declin|refund/i.test(e.event_type ?? '') ? 'warn' : 'info',\n ext_id: `${sourceId}:${e.provider_event_id ?? e.event_id}`.slice(0, 256),\n occurred_at: e.received_at,\n title: `${e.provider} ${e.event_type ?? 'event'}`.slice(0, 200),\n url: src.data.config?.link_url ?? null,\n summary: e.sig_verified === false ? 'unsigned delivery' : '',\n });\n if (res === 'created') out.written++;\n else if (res === 'duplicate') out.duplicates++;\n }\n\n const newest = fresh.length ? fresh[fresh.length - 1]!.event_id : null; // reversed: last = newest\n if (newest && (!watermark || newest > watermark)) {\n // advance the watermark \u2014 If-Match on the row's version: an overlapping\n // tick that already moved it wins, and the rows it wrote are 409s here\n const okAdv = await patchIntegration(base, H, src, {\n cursor: newest, status: 'ok', last_ok_at: new Date().toISOString(), consecutive_failures: 0,\n });\n if (okAdv) out.advanced++;\n }\n }\n return Response.json(out);\n },\n};\n\nasync function createFeed(base: string, H: Record<string, string>, row: FeedRow): Promise<'created' | 'duplicate' | 'failed'> {\n const r = await fetch(`${base}/v1/cms/items/events_feed`, {\n method: 'POST', headers: H, body: JSON.stringify({ status: 'published', data: row }),\n });\n if (r.ok) return 'created';\n return r.status === 409 ? 'duplicate' : 'failed';\n}\n\nasync function patchIntegration(base: string, H: Record<string, string>, row: Item<Integration>, data: Record<string, unknown>): Promise<boolean> {\n const r = await fetch(`${base}/v1/cms/items/integrations/${row.item_id}`, {\n method: 'PATCH',\n headers: row.version !== undefined ? { ...H, 'if-match': String(row.version) } : H,\n body: JSON.stringify({ data }),\n });\n return r.ok;\n}\n\nconst str = (v: unknown) => (typeof v === 'string' && v ? v : '');\n",
18477
+ "evaluate-alerts.ts": "// evaluate-alerts.ts \u2014 EVALUATE: rules \xD7 latest values \u2192 incidents, ONCE per\n// crossing (a vxil function; cron every 5 minutes).\n//\n// Each tick reads every enabled alert_rule (following next_cursor up to\n// MAX_RULES, and reporting `truncated` when there are more) and the\n// metric_latest rows they name (server mode: the restricted value is visible\n// here, never in a message). Per rule:\n//\n// breach = `op` against `threshold` \u2014 or, for op 'stale', no fresh value\n// for `threshold` minutes\n// ok \u2192 breach starts remember breach_since (no alert yet)\n// breach held for for_minutes ONE incident, created under a lock with a\n// guard (at most one open/acked incident per\n// rule), the rule flipped to `firing` (only once\n// an incident exists \u2014 a failed create leaves it\n// `ok` and the next tick retries), ONE chat\n// message, ONE mail+inbox message per on-call member\n// firing, still breached silence \u2014 a reminder every repeat_after_hours\n// firing \u2192 recovered the open incident resolved, ONE resolved message\n//\n// Every state change is a PATCH with If-Match on the rule's version, so two\n// overlapping ticks cannot both cross; the incident guard is the second fence.\n// A sensitive series never puts its number in a message or on the incident.\n//\n// cron-walk: complete-per-tick \u2014 every enabled rule is evaluated each tick (paged by next_cursor up to MAX_RULES, truncated reported)\n\nimport type { CronFunctionEnvelope } from '@vxil/sdk';\n\nconst MAX_RULES = 200;\nconst MAX_NOTIFY = 25; // on-call recipients per incident\nconst OPS_ORG_SLUG = 'ops';\nconst NOTIFY_ROLES = ['operator', 'ops-admin', 'owner'];\nconst DASHBOARD_URL = 'https://ops.example.com';\n\ntype Op = '>' | '<' | '>=' | '<=' | 'stale';\ninterface Rule {\n key: string; series: string; op: Op; threshold: number; for_minutes?: number; severity?: 'info' | 'warn' | 'crit';\n state?: 'ok' | 'firing'; enabled?: boolean; channels?: { chat?: boolean; email?: boolean };\n breach_since?: string | null; repeat_after_hours?: number; last_fired_at?: string | null; title?: string;\n}\ninterface Latest { series: string; value?: number | null; restricted_value?: number | null; sensitive?: boolean; updated_at?: string; status?: string; label?: string; unit?: string }\ninterface Incident { rule: string; state: string; severity?: string; title?: string }\ninterface Item<T> { item_id: string; version?: number | undefined; data: T }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as CronFunctionEnvelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cmsJwt = env.scoped_jwts?.cms;\n if (!cmsJwt) return Response.json({ error: 'missing cms scope' }, { status: 403 });\n const H = { authorization: `Bearer ${cmsJwt}`, 'content-type': 'application/json' };\n const now = new Date();\n const nowIso = now.toISOString();\n\n // 1. every enabled rule (complete per tick, capped)\n const rules: Item<Rule>[] = [];\n let cursor: string | null = null;\n let truncated = false;\n do {\n const q = new URLSearchParams({ filter: JSON.stringify({ enabled: true }), limit: '100' });\n if (cursor) q.set('cursor', cursor);\n const r = await fetch(`${base}/v1/cms/items/alert_rules?${q}`, { headers: H });\n if (!r.ok) return Response.json({ error: 'rules_read_failed', status: r.status }, { status: 502 });\n const d = ((await r.json()) as { data?: { items?: Item<Rule>[]; next_cursor?: string | null } }).data;\n rules.push(...(d?.items ?? []));\n cursor = d?.next_cursor ?? null;\n if (rules.length >= MAX_RULES) { truncated = cursor !== null; break; }\n } while (cursor);\n\n // 2. the latest value of every series they name ($in takes 50 values)\n const latest = new Map<string, Item<Latest>>();\n const names = [...new Set(rules.map((r) => r.data.series))];\n for (let i = 0; i < names.length; i += 50) {\n const q = new URLSearchParams({ filter: JSON.stringify({ series: { $in: names.slice(i, i + 50) } }), limit: '100' });\n const r = await fetch(`${base}/v1/cms/items/metric_latest?${q}`, { headers: H });\n if (!r.ok) continue;\n for (const it of ((await r.json()) as { data?: { items?: Item<Latest>[] } }).data?.items ?? []) latest.set(it.data.series, it);\n }\n\n const chat = env.secrets?.slack_webhook_url;\n const out = { rules: rules.length, fired: 0, resolved: 0, reminded: 0, pending: 0, unknown: 0, failed: 0, posted: 0, notified: 0, truncated };\n let recipients: string[] | null = null; // resolved once per tick, only when something fires\n\n for (const rule of rules) {\n const r = rule.data;\n const l = latest.get(r.series);\n const verdict = breached(r, l?.data, now);\n if (verdict === 'unknown') { out.unknown++; continue; } // no data yet for a threshold rule: could not check\n const sensitive = Boolean(l?.data.sensitive);\n const shown = sensitive ? null : (l?.data.value ?? null);\n const title = r.title ?? `${r.series} ${r.op} ${r.threshold}`;\n const state = r.state ?? 'ok';\n\n if (verdict && state === 'ok') {\n if (!r.breach_since) {\n await patchRule(base, H, rule, { breach_since: nowIso, last_eval_at: nowIso });\n if ((r.for_minutes ?? 0) > 0) { out.pending++; continue; }\n } else if (now.getTime() - Date.parse(r.breach_since) < (r.for_minutes ?? 0) * 60_000) {\n out.pending++;\n continue;\n }\n // THE CROSSING. The incident first: its lock + guard admit one open\n // incident per rule, so of two racing ticks exactly one gets a 201.\n const inc = await fetch(`${base}/v1/cms/items/incidents`, {\n method: 'POST', headers: H,\n body: JSON.stringify({\n status: 'published',\n lock: `incident:${r.key}`,\n guard: { filter: { rule: r.key, state: { $in: ['open', 'acked'] } }, max: 1 },\n data: { rule: r.key, state: 'open', severity: r.severity ?? 'warn', opened_at: nowIso, title, last_value: shown },\n }),\n });\n // The rule flips to `firing` only when an incident exists: a 201 (ours) or a\n // 409 guard_failed (one is already open). Any other answer \u2014 a 5xx, a hook's\n // 422, a 409 that is not the guard \u2014 leaves the rule `ok` with its\n // breach_since, so the next tick crosses again instead of losing the alert.\n const guardHeld = !inc.ok && inc.status === 409\n && ((await inc.clone().json().catch(() => ({}))) as { error?: { code?: string } }).error?.code === 'guard_failed';\n if (!inc.ok && !guardHeld) { out.failed++; continue; }\n // fresh version: patchRule above may have bumped it\n await patchRule(base, H, rule, { state: 'firing', last_fired_at: nowIso, last_eval_at: nowIso }, true);\n if (!inc.ok) continue; // guard_failed: an incident is already open \u2014 no second message\n out.fired++;\n await markTile(base, H, l, r.op === 'stale' ? 'stale' : r.severity === 'crit' ? 'crit' : 'warn');\n const value = shown === null ? '' : ` (now ${shown}${l?.data.unit ? ` ${l.data.unit}` : ''})`;\n if (r.channels?.chat !== false && chat) {\n if ((await post(chat, `${dot(r.severity)} [${(r.severity ?? 'warn').toUpperCase()}] ${title}${value} \u2014 ${DASHBOARD_URL}`)).ok) out.posted++;\n }\n if (r.channels?.email) {\n recipients ??= await onCall(base, env.scoped_jwts?.orgs);\n const incId = ((await inc.json().catch(() => ({}))) as { data?: { item_id?: string } }).data?.item_id ?? r.key;\n out.notified += await notify(base, env.scoped_jwts?.notifications, recipients, incId, title, value);\n }\n continue;\n }\n\n if (verdict && state === 'firing') {\n const every = r.repeat_after_hours ?? 0;\n const last = Date.parse(r.last_fired_at ?? '') || 0;\n if (every > 0 && now.getTime() - last >= every * 3600_000) {\n if (await patchRule(base, H, rule, { last_fired_at: nowIso, last_eval_at: nowIso }) && chat && r.channels?.chat !== false) {\n out.reminded++;\n if ((await post(chat, `\u23F0 STILL FIRING: ${title} (since ${r.breach_since ?? r.last_fired_at ?? '?'})`)).ok) out.posted++;\n }\n }\n continue;\n }\n\n if (!verdict && state === 'firing') {\n // RECOVERY \u2014 the If-Match on the rule picks the one tick that resolves\n if (!(await patchRule(base, H, rule, { state: 'ok', breach_since: null, last_eval_at: nowIso }))) continue;\n const q = new URLSearchParams({ filter: JSON.stringify({ rule: r.key, state: { $in: ['open', 'acked'] } }), limit: '5' });\n const open = await fetch(`${base}/v1/cms/items/incidents?${q}`, { headers: H });\n for (const inc of ((await open.json().catch(() => ({}))) as { data?: { items?: Item<Incident>[] } }).data?.items ?? []) {\n await fetch(`${base}/v1/cms/items/incidents/${inc.item_id}`, {\n method: 'PATCH', headers: H,\n body: JSON.stringify({ data: { state: 'resolved', resolved_at: nowIso }, if: { state: inc.data.state } }),\n });\n }\n out.resolved++;\n await markTile(base, H, l, 'ok');\n if (chat && r.channels?.chat !== false && (await post(chat, `\u{1F7E2} RESOLVED: ${title}`)).ok) out.posted++;\n continue;\n }\n\n if (!verdict && r.breach_since) {\n await patchRule(base, H, rule, { breach_since: null, last_eval_at: nowIso }); // cleared before it was sustained\n }\n }\n return Response.json(out);\n },\n};\n\n/** true = breached, false = fine, 'unknown' = no row for the series yet (or no value). */\nfunction breached(r: Rule, l: Latest | undefined, now: Date): boolean | 'unknown' {\n if (r.op === 'stale') {\n if (!l) return 'unknown'; // never reported: a missing source, not a stale one\n const at = Date.parse(l.updated_at ?? '');\n return !Number.isFinite(at) || now.getTime() - at > r.threshold * 60_000;\n }\n const v = l ? (l.sensitive ? l.restricted_value : l.value) : null;\n if (typeof v !== 'number') return 'unknown';\n switch (r.op) {\n case '>': return v > r.threshold;\n case '<': return v < r.threshold;\n case '>=': return v >= r.threshold;\n case '<=': return v <= r.threshold;\n default: return 'unknown';\n }\n}\n\n/** PATCH a rule with If-Match; `fresh` re-reads the version first. False on a 409. */\nasync function patchRule(base: string, H: Record<string, string>, rule: Item<Rule>, data: Partial<Rule> & Record<string, unknown>, fresh = false): Promise<boolean> {\n if (fresh) {\n const g = await fetch(`${base}/v1/cms/items/alert_rules/${rule.item_id}`, { headers: H });\n if (g.ok) rule.version = ((await g.json()) as { data?: { version?: number } }).data?.version ?? rule.version;\n }\n const res = await fetch(`${base}/v1/cms/items/alert_rules/${rule.item_id}`, {\n method: 'PATCH',\n headers: rule.version !== undefined ? { ...H, 'if-match': String(rule.version) } : H,\n body: JSON.stringify({ data }),\n });\n if (res.ok) {\n rule.version = ((await res.json().catch(() => ({}))) as { data?: { version?: number } }).data?.version ?? (rule.version ?? 0) + 1;\n Object.assign(rule.data, data);\n }\n return res.ok;\n}\n\n/** The tile's colour follows the rule; a crossing is the only write. */\nasync function markTile(base: string, H: Record<string, string>, l: Item<Latest> | undefined, status: string): Promise<void> {\n if (!l || l.data.status === status) return;\n await fetch(`${base}/v1/cms/items/metric_latest/${l.item_id}`, {\n method: 'PATCH', headers: H, body: JSON.stringify({ data: { status } }),\n });\n}\n\n/** Members of the ops org holding an on-call role. */\nasync function onCall(base: string, orgsJwt: string | undefined): Promise<string[]> {\n if (!orgsJwt) return [];\n const H = { authorization: `Bearer ${orgsJwt}` };\n const orgs = await fetch(`${base}/v1/orgs`, { headers: H });\n const org = ((await orgs.json().catch(() => ({}))) as { data?: { orgs?: Array<{ org_id: string; slug: string }> } }).data?.orgs\n ?.find((o) => o.slug === OPS_ORG_SLUG);\n if (!org) return [];\n const m = await fetch(`${base}/v1/orgs/${encodeURIComponent(org.org_id)}/members`, { headers: H });\n const members = ((await m.json().catch(() => ({}))) as { data?: { members?: Array<{ user_id: string; role: string }> } }).data?.members ?? [];\n return members.filter((x) => NOTIFY_ROLES.includes(x.role)).map((x) => x.user_id).slice(0, MAX_NOTIFY);\n}\n\n/** One transactional e-mail + inbox message per recipient; the key makes a re-run a no-op. */\nasync function notify(base: string, jwt: string | undefined, users: string[], incidentId: string, title: string, value: string): Promise<number> {\n if (!jwt) return 0;\n let sent = 0;\n for (const user_id of users) {\n const r = await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: { authorization: `Bearer ${jwt}`, 'content-type': 'application/json', 'idempotency-key': `incident:${incidentId}:${user_id}` },\n body: JSON.stringify({\n user_id, template: 'transactional', channel: 'both',\n data: { subject: `Alert: ${title}`, paragraph: `${title}${value}. Open the dashboard to acknowledge it.`, cta_label: 'Open the dashboard', cta_url: DASHBOARD_URL },\n }),\n });\n if (r.ok) sent++;\n }\n return sent;\n}\n\nconst dot = (s?: string) => (s === 'crit' ? '\u{1F534}' : s === 'warn' ? '\u{1F7E0}' : '\u{1F535}');\n\n// \u2500\u2500 the chat webhook: `text` is Slack's field, `content` is Discord's \u2014 send both,\n// EXCEPT to a Google Chat space webhook (chat.googleapis.com), which takes\n// `{ text }` alone \u2014 an unknown field can be rejected there, so it is dropped.\nasync function post(url: string, text: string): Promise<{ ok: boolean; status: number }> {\n let host = '';\n try { host = new URL(url).hostname; } catch { /* unparseable \u2192 the generic body below */ }\n const body = host === 'chat.googleapis.com' ? { text } : { text, content: text };\n const res = await fetch(url, {\n method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body),\n }).catch(() => null);\n return { ok: Boolean(res?.ok), status: res?.status ?? 0 };\n}\n",
18478
+ "ops-action.ts": "// ops-action.ts \u2014 the dashboard's ONLY write door (a vxil function, http).\n//\n// The browser key is read-only (cms:read, no cms:write), so every staff write\n// arrives here as `POST /v1/fn/ops-action { op, \u2026 }` with the signed-in\n// person's session. Each op names the org permission it needs, and the\n// function asks orgs \u2014 the revocation-grade check, not the session snapshot \u2014\n// before it writes:\n//\n// ack | resolve incidents.manage compare-and-set on the state\n// upsert_rule | toggle_rule rules.manage the series must exist\n// update_integration | pause_integration integrations.manage\n// non-secret settings only\n//\n// A refusal is `403 { error: 'role_required', permission }`. The writes are\n// audited as the project (the function writes with the project's cms scope);\n// the person is recorded in the row itself \u2014 `acked_by` on an incident \u2014 and\n// was checked against orgs before anything was written.\n//\n// Integration settings never hold a credential: any key that looks like one\n// (key, token, secret, password, authorization) is stripped before the write\n// and named in the answer. Credentials are project secrets, set with\n// `vxil secrets set`, never from a browser. And the settings that decide WHERE\n// a credential goes \u2014 config.base_url, config.auth, config.peer \u2014 are not\n// editable here at all (409 locked_config): they change only with a server key\n// (`vxil api PATCH \u2026` or a re-seed), and collect.ts binds each secret to one host\n// in code besides.\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\nconst OPS_ORG_SLUG = 'ops';\nconst CREDENTIAL_KEY = /key|token|secret|password|passwd|authorization|credential/i;\nconst CREDENTIAL_VALUE = /^(?:bearer\\s|sk_|rk_|pk_live_|ghp_|xox[abp]-)/i;\n// where a credential goes: never changed from the browser\nconst LOCKED_CONFIG = ['base_url', 'auth', 'peer'] as const;\n\ntype Payload = {\n op?: string;\n incident_id?: string; note?: string;\n rule?: { key?: string; series?: string; op?: string; threshold?: number; for_minutes?: number; severity?: string; repeat_after_hours?: number; channels?: { chat?: boolean; email?: boolean }; enabled?: boolean; title?: string };\n key?: string; enabled?: boolean; paused?: boolean;\n integration?: { key?: string; label?: string; owner?: string; poll_every_min?: number; config?: Record<string, unknown> };\n};\ntype Env = HttpFunctionEnvelope<Payload>;\ninterface Item<T> { item_id: string; version?: number; data: T }\n\nconst PERMISSION: Record<string, string> = {\n ack: 'incidents.manage',\n resolve: 'incidents.manage',\n upsert_rule: 'rules.manage',\n toggle_rule: 'rules.manage',\n update_integration: 'integrations.manage',\n pause_integration: 'integrations.manage',\n};\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cmsJwt = env.scoped_jwts?.cms;\n if (!cmsJwt) return json({ error: 'missing cms scope' }, 403);\n const user = env.end_user;\n if (!user) return json({ error: 'sign_in_required' }, 401);\n const p = env.payload ?? {};\n const permission = PERMISSION[p.op ?? ''];\n if (!permission) return json({ error: 'unknown_op', ops: Object.keys(PERMISSION) }, 400);\n if (!(await orgAllows(base, env.scoped_jwts?.orgs, user.id, permission))) {\n return json({ error: 'role_required', permission }, 403);\n }\n const H = { authorization: `Bearer ${cmsJwt}`, 'content-type': 'application/json' };\n const now = new Date().toISOString();\n\n switch (p.op) {\n case 'ack':\n case 'resolve': {\n if (!p.incident_id) return json({ error: 'incident_id required' }, 400);\n const data = p.op === 'ack'\n ? { state: 'acked', acked_by: user.id, ...(p.note ? { note: p.note.slice(0, 2000) } : {}) }\n : { state: 'resolved', resolved_at: now, ...(p.note ? { note: p.note.slice(0, 2000) } : {}) };\n // compare-and-set: only from the state the button was showing\n const from = p.op === 'ack' ? ['open'] : ['open', 'acked'];\n const r = await fetch(`${base}/v1/cms/items/incidents/${encodeURIComponent(p.incident_id)}`, {\n method: 'PATCH', headers: H, body: JSON.stringify({ data, if: { state: { $in: from } } }),\n });\n if (r.status === 409) return json({ error: 'state_changed', hint: 'someone else moved this incident; reload it' }, 409);\n if (!r.ok) return json({ error: 'write_failed', status: r.status }, r.status === 404 ? 404 : 502);\n return json({ ok: true, op: p.op, incident_id: p.incident_id, state: data.state }, 200);\n }\n\n case 'upsert_rule': {\n const r = p.rule ?? {};\n if (!r.key || !r.series || !r.op || typeof r.threshold !== 'number') {\n return json({ error: 'rule needs key, series, op and a numeric threshold' }, 400);\n }\n // the series must be one the dashboard already collects\n const seen = await list(base, H, 'metric_latest', { series: r.series }, 1);\n if (seen.length === 0) return json({ error: 'unknown_series', series: r.series }, 422);\n const data: Record<string, unknown> = {\n key: r.key, series: r.series, op: r.op, threshold: r.threshold,\n for_minutes: r.for_minutes ?? 0, severity: r.severity ?? 'warn', repeat_after_hours: r.repeat_after_hours ?? 24,\n channels: { chat: r.channels?.chat !== false, email: Boolean(r.channels?.email) },\n enabled: r.enabled !== false, title: (r.title ?? `${r.series} ${r.op} ${r.threshold}`).slice(0, 200),\n };\n const existing = (await list<{ key: string }>(base, H, 'alert_rules', { key: r.key }, 1))[0];\n const w = existing\n ? await fetch(`${base}/v1/cms/items/alert_rules/${existing.item_id}`, { method: 'PATCH', headers: H, body: JSON.stringify({ data }) })\n : await fetch(`${base}/v1/cms/items/alert_rules`, {\n method: 'POST', headers: H, body: JSON.stringify({ status: 'published', data: { ...data, state: 'ok' } }),\n });\n if (!w.ok) return json({ error: 'write_failed', status: w.status, detail: await errorCode(w) }, w.status === 422 ? 422 : 502);\n return json({ ok: true, op: p.op, key: r.key, created: !existing }, existing ? 200 : 201);\n }\n\n case 'toggle_rule': {\n if (!p.key || typeof p.enabled !== 'boolean') return json({ error: 'key and enabled required' }, 400);\n const row = (await list(base, H, 'alert_rules', { key: p.key }, 1))[0];\n if (!row) return json({ error: 'unknown_rule' }, 404);\n // switching a rule off also clears a pending breach; a firing rule's\n // open incident stays for a human to resolve\n const w = await fetch(`${base}/v1/cms/items/alert_rules/${row.item_id}`, {\n method: 'PATCH', headers: H, body: JSON.stringify({ data: { enabled: p.enabled, ...(p.enabled ? {} : { breach_since: null }) } }),\n });\n return w.ok ? json({ ok: true, op: p.op, key: p.key, enabled: p.enabled }, 200) : json({ error: 'write_failed', status: w.status }, 502);\n }\n\n case 'update_integration': {\n const i = p.integration ?? {};\n if (!i.key) return json({ error: 'integration.key required' }, 400);\n const row = (await list<{ config?: Record<string, unknown> }>(base, H, 'integrations', { key: i.key }, 1))[0];\n if (!row) return json({ error: 'unknown_integration' }, 404);\n const stripped: string[] = [];\n const data: Record<string, unknown> = {};\n if (typeof i.label === 'string') data.label = i.label.slice(0, 120);\n if (typeof i.owner === 'string') data.owner = i.owner.slice(0, 120);\n if (typeof i.poll_every_min === 'number') data.poll_every_min = Math.max(1, Math.min(1440, Math.round(i.poll_every_min)));\n if (i.config && typeof i.config === 'object' && !Array.isArray(i.config)) {\n // base_url / auth / peer decide where a credential is sent: refuse a change,\n // and carry the stored values over (a config without `auth` must not\n // silently turn a public source into one that sends a token)\n const stored = row.data.config ?? {};\n const locked = LOCKED_CONFIG.filter((k) => k in i.config! && JSON.stringify(i.config![k]) !== JSON.stringify(stored[k]));\n if (locked.length > 0) {\n return json({ error: 'locked_config', fields: locked, hint: 'base_url, auth and peer change only with a server key' }, 409);\n }\n const next = stripCredentials(i.config, '', stripped);\n for (const k of LOCKED_CONFIG) { if (k in stored) next[k] = stored[k]; else delete next[k]; }\n data.config = next;\n }\n if (Object.keys(data).length === 0) return json({ error: 'nothing to update', stripped }, 400);\n const w = await fetch(`${base}/v1/cms/items/integrations/${row.item_id}`, {\n method: 'PATCH', headers: H, body: JSON.stringify({ data }),\n });\n if (!w.ok) return json({ error: 'write_failed', status: w.status }, 502);\n return json({ ok: true, op: p.op, key: i.key, stripped }, 200);\n }\n\n case 'pause_integration': {\n if (!p.key || typeof p.paused !== 'boolean') return json({ error: 'key and paused required' }, 400);\n const row = (await list(base, H, 'integrations', { key: p.key }, 1))[0];\n if (!row) return json({ error: 'unknown_integration' }, 404);\n const data = p.paused\n ? { enabled: false, status: 'paused' }\n : { enabled: true, status: 'ok', next_due_at: now, consecutive_failures: 0 }; // resumed: polled on the next tick\n const w = await fetch(`${base}/v1/cms/items/integrations/${row.item_id}`, {\n method: 'PATCH', headers: H, body: JSON.stringify({ data }),\n });\n return w.ok ? json({ ok: true, op: p.op, key: p.key, paused: p.paused }, 200) : json({ error: 'write_failed', status: w.status }, 502);\n }\n }\n return json({ error: 'unknown_op' }, 400);\n },\n};\n\n/** Drop every key (at any depth) that looks like a credential, and any value\n * that looks like a bearer token or a provider secret. Names what it dropped. */\nfunction stripCredentials(v: Record<string, unknown>, path: string, stripped: string[]): Record<string, unknown> {\n const out: Record<string, unknown> = {};\n for (const [k, val] of Object.entries(v)) {\n const at = path ? `${path}.${k}` : k;\n if (CREDENTIAL_KEY.test(k) || (typeof val === 'string' && CREDENTIAL_VALUE.test(val))) { stripped.push(at); continue; }\n if (Array.isArray(val)) {\n out[k] = val.map((x, i) => (x && typeof x === 'object' && !Array.isArray(x) ? stripCredentials(x as Record<string, unknown>, `${at}.${i}`, stripped) : x));\n } else if (val && typeof val === 'object') {\n out[k] = stripCredentials(val as Record<string, unknown>, at, stripped);\n } else {\n out[k] = val;\n }\n }\n return out;\n}\n\nasync function list<T = Record<string, unknown>>(base: string, H: Record<string, string>, coll: string, filter: Record<string, unknown>, limit: number): Promise<Item<T>[]> {\n const q = new URLSearchParams({ filter: JSON.stringify(filter), limit: String(limit) });\n const r = await fetch(`${base}/v1/cms/items/${coll}?${q}`, { headers: H });\n if (!r.ok) return [];\n return ((await r.json()) as { data?: { items?: Item<T>[] } }).data?.items ?? [];\n}\n\nasync function errorCode(r: Response): Promise<string | undefined> {\n return ((await r.json().catch(() => ({}))) as { error?: { code?: string; message?: string } }).error?.message;\n}\n\n/** Revocation-grade check: does this user hold `permission` in the ops org? */\nasync function orgAllows(base: string, orgsJwt: string | undefined, userId: string, permission: string): Promise<boolean> {\n if (!orgsJwt) return false;\n const H = { authorization: `Bearer ${orgsJwt}` };\n const mine = await fetch(`${base}/v1/orgs?user_id=${encodeURIComponent(userId)}`, { headers: H });\n if (!mine.ok) return false;\n const org = ((await mine.json()) as { data?: { orgs?: Array<{ org_id: string; slug: string }> } }).data?.orgs\n ?.find((o) => o.slug === OPS_ORG_SLUG);\n if (!org) return false;\n const q = `user_id=${encodeURIComponent(userId)}&permission=${encodeURIComponent(permission)}`;\n const check = await fetch(`${base}/v1/orgs/${encodeURIComponent(org.org_id)}/check?${q}`, { headers: H });\n if (!check.ok) return false;\n return ((await check.json()) as { data?: { allowed?: boolean } }).data?.allowed === true;\n}\n\nconst json = (o: unknown, status: number) => Response.json(o, { status });\n"
18479
+ }
18480
+ },
18481
+ {
18482
+ "id": "service-monitor",
18483
+ "title": "Service monitor",
18484
+ "vertical": "ops",
18485
+ "summary": "Uptime and health checks for your team's public endpoints plus dead-man's-switch heartbeats from your own jobs: one cron function probes through a deny-by-default egress allowlist, opens and closes an outage once per state crossing (lock + guard), pages the on-call shift from data in chat, e-mail and the inbox, and counts hourly uptime with atomic increments. Staff acknowledge, resolve and post updates through one permission-checked function, and a keyless public status page reads only the components and updates you publish.",
18486
+ "collections": [
18487
+ "checks",
18488
+ "uptime_hourly",
18489
+ "outages",
18490
+ "outage_updates",
18491
+ "status_components",
18492
+ "oncall_shifts"
18493
+ ],
18494
+ "features": [
18495
+ "auth",
18496
+ "orgs",
18497
+ "cms",
18498
+ "notifications",
18499
+ "functions"
18500
+ ],
18501
+ "hasFunctions": true,
18502
+ "byoKeys": [
18503
+ "slack_webhook_url",
18504
+ "probe_token",
18505
+ "oidc_client_secret"
18506
+ ],
18507
+ "configSrc": "import { defineConfig } from '@vxil/config';\n\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n// \"Service monitor\" \u2014 uptime and health checks for your team's PUBLIC\n// endpoints, dead-man's-switch heartbeats from your own jobs, outages that open\n// and close once per state crossing, the on-call shift paged from data, and a\n// keyless public status page. Declared end-to-end in ONE typed file.\n//\n// \u2022 auth \u2192 staff sign in through your identity provider (OIDC);\n// password and magic-link sign-in are off \u2014 staff only\n// \u2022 orgs \u2192 one team org; the roles `oncall` (outages.manage) and\n// `monitor-admin` (checks.manage) are rows you add once\n// \u2022 cms \u2192 checks \xB7 uptime_hourly \xB7 outages \xB7 oncall_shifts (private)\n// and status_components \xB7 outage_updates (\u2B22 PUBLIC: only\n// their PUBLISHED rows are readable with no key)\n// \u2022 notifications \u2192 the page: e-mail + the in-app inbox of the shift\n// \u2022 functions \u2192 probe (cron) \xB7 heartbeat (http) \xB7 outage-action (http) \xB7\n// daily-report (cron)\n//\n// There is no monitoring engine, no escalation engine and no connector here:\n// a probe is a cron function behind a deny-by-default egress allowlist, an\n// outage is a row whose creation is guarded, a page is a read of who is on\n// shift right now, and escalation is one more filter in the same cron.\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\nexport default defineConfig({\n env: 'staging',\n\n features: {\n // \u2500\u2500 Staff-only sign-in \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n // The OIDC issuer is your company's identity provider; `allowedDomains`\n // fences sign-in to your own e-mail domain (no e-mail \u21D2 refused). Password\n // and magic-link sign-in are switched off, so nobody outside the IdP can\n // create an account. Store the client secret once:\n // printf '%s' \"$OIDC_SECRET\" | vxil secrets set auth/oidc_client_secret\n auth: {\n methods: { emailPassword: false, magicLink: false },\n providers: {\n oidc: {\n issuer: 'https://login.example-idp.com', // https, no query/fragment\n clientId: 'vxil-service-monitor', // not a secret: it rides every authorize URL\n clientSecretRef: 'secret:oidc_client_secret', // the `secrets` block below\n scopes: ['email', 'profile'], // `openid` is always added\n allowedDomains: ['example.com'], // fail-closed domain fence\n },\n },\n // The member's org role rides the session (a snapshot, refreshed with it).\n // The functions below do NOT trust it: they re-check the permission live.\n orgClaims: { enabled: true },\n session: { ttlMinutes: 60, refreshTtlDays: 7 },\n security: {\n // every sign-in return URL must be your own admin app\n allowedRedirectOrigins: ['https://status-admin.example.com'],\n },\n },\n\n // \u2500\u2500 One team org, three roles \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n // `viewer` is built in (reads only). `oncall` and `monitor-admin` are\n // CUSTOM roles \u2014 rows, not config \u2014 added once with `POST /v1/orgs/roles`\n // (see the README). `owner` holds every permission.\n orgs: { enabled: true, maxOrgsPerTenant: 5, maxMembersPerOrg: 500, invitationTtlHours: 72 },\n\n cms: {\n // Draft \u2192 publish is ON, and it matters for exactly one collection:\n // `outage_updates`. A staff note is written as a DRAFT and is never\n // public; only an update the team chooses to publish reaches the status\n // page. Every other write in this blueprint passes `status` explicitly.\n draftPublish: true,\n hooks: {\n // The OUTAGE LIFECYCLE, enforced in the same write: open \u2192 acked \u2192\n // resolved, open \u2192 resolved; `resolved` is terminal.\n outage_stage: {\n collection: 'outages',\n event: 'beforeUpdate',\n kind: 'validate',\n expr:\n 'item.state == before.state'\n + \" || (before.state == 'open' && (item.state == 'acked' || item.state == 'resolved'))\"\n + \" || (before.state == 'acked' && item.state == 'resolved')\",\n message: 'illegal outage state transition (open \u2192 acked \u2192 resolved)',\n },\n // ONE uptime row per check per hour: the composed key is DERIVED\n // server-side, and `point_key` is unique \u2014 a racing second create is\n // a 409, never a second row. (Hooks do not run on $inc, which is fine:\n // an increment never changes check or bucket.)\n uptime_key: {\n collection: 'uptime_hourly',\n event: 'beforeWrite',\n kind: 'derive',\n field: 'point_key',\n expr: \"concat(item.check, '|', item.bucket)\",\n },\n check_kind: {\n collection: 'checks',\n event: 'beforeWrite',\n kind: 'validate',\n expr: \"item.kind == 'http' || item.kind == 'keyword' || item.kind == 'heartbeat'\",\n message: 'kind must be http, keyword or heartbeat',\n },\n // A probed URL is https, always. A heartbeat check has no URL.\n check_url: {\n collection: 'checks',\n event: 'beforeWrite',\n kind: 'validate',\n expr: \"item.kind == 'heartbeat' || startsWith(item.url, 'https://')\",\n message: 'an http or keyword check needs an https:// url',\n },\n },\n // OPTIONAL live view for your admin screen: every outage write is pushed to\n // a realtime channel (enable the realtime feature too). Safe with\n // `payload: 'full'` because no `outages` field declares readRoles.\n // cdc: {\n // outages_live: { collection: 'outages', channel: 'monitor:outages', payload: 'full' },\n // },\n },\n\n // The page goes out as the shipped `transactional` message (a subject + a\n // paragraph + a button to the outage) on e-mail AND the in-app inbox.\n notifications: {\n provider: 'mock', // switch to your e-mail provider before production\n fromEmail: 'monitor@example.com',\n fromName: 'Service monitor',\n inboxEnabled: true,\n },\n\n functions: { enabled: true },\n },\n\n // \u2500\u2500 Schema-as-code (\u22648 index slots per collection: s1\u2013s4/n1\u2013n2/t1\u2013t2) \u2500\u2500\u2500\u2500\u2500\u2500\n cms: {\n collections: {\n // What to watch. One row per endpoint or heartbeat.\n checks: {\n singular: 'check',\n fields: {\n key: { type: 'string', required: true, indexSlot: 's1', unique: true }, // 'api-health'\n kind: { type: 'string', required: true, indexSlot: 's2' }, // http | keyword | heartbeat (hook-checked)\n component: { type: 'string', indexSlot: 's3' }, // a status_components key\n state: { type: 'string', indexSlot: 's4', validation: { enum: ['up', 'degraded', 'down', 'paused'] } },\n consecutive_failures: { type: 'int', indexSlot: 'n1', validation: { min: 0 } },\n last_latency_ms: { type: 'float', indexSlot: 'n2' },\n next_due_at: { type: 'datetime', indexSlot: 't1' }, // the probe's work-queue filter\n last_checked_at: { type: 'datetime', indexSlot: 't2' },\n // \u2500\u2500 unslotted \u2500\u2500\n name: { type: 'string', validation: { max: 120 } },\n url: { type: 'string', validation: { max: 2000 } }, // https only (check_url hook)\n expect_status: { type: 'int', validation: { min: 100, max: 599 } }, // default 200\n expect_contains: { type: 'string', validation: { max: 200 } }, // keyword checks\n send_probe_token: { type: 'bool' }, // send `x-probe-token: <probe_token secret>`\n timeout_ms: { type: 'int', validation: { min: 1000, max: 15000 } }, // default 10000\n degraded_ms: { type: 'int', validation: { min: 1 } }, // slower than this \u21D2 degraded\n interval_min: { type: 'int', validation: { min: 1, max: 1440 } }, // default 1\n failures_to_down: { type: 'int', validation: { min: 1, max: 10 } }, // default 2\n enabled: { type: 'bool', indexed: true },\n heartbeat_grace_min: { type: 'int', validation: { min: 1, max: 10080 } }, // heartbeat checks\n last_heartbeat_at: { type: 'datetime' },\n hour_bucket: { type: 'string' }, // 'YYYY-MM-DDTHH' of hour_row\n hour_row: { type: 'string' }, // the current uptime_hourly item id\n hour_max: { type: 'float' }, // the slowest sample this hour (drives latency_max)\n open_outage: { type: 'string' }, // the outage this down-crossing opened\n last_error: { type: 'text' },\n },\n },\n\n // Hourly counters: ok / total per check per hour, incremented atomically.\n uptime_hourly: {\n singular: 'uptime_point',\n fields: {\n check: { type: 'string', required: true, indexSlot: 's1' }, // a checks key\n bucket: { type: 'string', required: true, indexSlot: 's2' }, // 'YYYY-MM-DDTHH' (UTC)\n point_key: { type: 'string', indexSlot: 's3', unique: true }, // derived: check|bucket\n ok: { type: 'int', indexSlot: 'n1', validation: { min: 0 } },\n total: { type: 'int', indexSlot: 'n2', validation: { min: 0 } },\n ts: { type: 'datetime', indexSlot: 't1' }, // the hour's start \u2014 the report window field\n latency_sum: { type: 'float' },\n latency_max: { type: 'float' },\n },\n },\n\n // One row per down-crossing. Created under a lock + guard, so a check can\n // never have two open outages, however often a tick is re-delivered.\n outages: {\n singular: 'outage',\n fields: {\n check: { type: 'string', required: true, indexSlot: 's1' },\n state: { type: 'string', indexSlot: 's2', validation: { enum: ['open', 'acked', 'resolved'] } },\n severity: { type: 'string', indexSlot: 's3', validation: { enum: ['minor', 'major'] } },\n acked_by: { type: 'string', indexSlot: 's4' }, // the acknowledging staff member\n opened_at: { type: 'datetime', indexSlot: 't1' },\n resolved_at: { type: 'datetime', indexSlot: 't2' },\n title: { type: 'string', validation: { max: 200 } },\n cause: { type: 'text' },\n duration_min: { type: 'float' },\n paged: { type: 'json' }, // who was paged, when \u2014 written once (null \u21D2 not yet)\n escalated_at: { type: 'datetime' }, // the secondary was paged\n },\n },\n\n // \u2B22 PUBLIC: the updates the team PUBLISHES. A note stays a draft \u2014 and a\n // draft is never served on the public lane.\n outage_updates: {\n singular: 'outage_update',\n public: true,\n fields: {\n outage: { type: 'relation', relationTo: 'outages', required: true, indexSlot: 's1' },\n kind: {\n type: 'string', required: true, indexSlot: 's2',\n validation: { enum: ['investigating', 'identified', 'monitoring', 'resolved', 'note'] },\n },\n // the author's user id: visible to signed-in staff with a role, never on\n // the public lane (a readRoles field is never served anonymously)\n author: { type: 'string', indexSlot: 's3', readRoles: ['owner', 'admin', 'viewer', 'oncall', 'monitor-admin'] },\n at: { type: 'datetime', indexSlot: 't1' },\n component: { type: 'string' }, // the status_components key it is about\n body: { type: 'text', validation: { max: 2000 } },\n },\n },\n\n // \u2B22 PUBLIC: the components on your status page and their current status.\n status_components: {\n singular: 'status_component',\n public: true,\n fields: {\n key: { type: 'string', required: true, indexSlot: 's1', unique: true }, // 'api'\n name: { type: 'string', required: true, indexSlot: 's2' }, // 'API'\n group: { type: 'string', indexSlot: 's3' },\n status: {\n type: 'string', indexSlot: 's4',\n validation: { enum: ['operational', 'degraded', 'partial_outage', 'major_outage'] },\n },\n sort: { type: 'int', indexSlot: 'n1' },\n updated_at: { type: 'datetime', indexSlot: 't1' },\n description: { type: 'text' },\n },\n },\n\n // Who is on call, as DATA. The probe pages whoever's shift covers now.\n oncall_shifts: {\n singular: 'oncall_shift',\n fields: {\n user: { type: 'string', required: true, indexSlot: 's1' }, // the staff member's user id\n email: { type: 'string', indexSlot: 's2' }, // optional delivery override\n role: { type: 'string', indexSlot: 's3', validation: { enum: ['primary', 'secondary'] } },\n starts_at: { type: 'datetime', required: true, indexSlot: 't1' },\n ends_at: { type: 'datetime', required: true, indexSlot: 't2' },\n },\n },\n },\n },\n\n functions: {\n // THE PROBE. Every minute: due checks \u2192 probe in parallel \u2192 counters \u2192\n // once-per-crossing outages, component status and pages \u2192 escalation.\n probe: {\n entry: './functions/probe.ts',\n // Free plan: '*/15 * * * *' (its fastest function cron). `overlap: 'skip'`:\n // a tick still running never runs beside the next one.\n trigger: { kind: 'cron', schedule: '* * * * *', overlap: 'skip' },\n scopes: ['cms:read', 'cms:write', 'notifications:send'],\n secrets: ['secret:slack_webhook_url', 'secret:probe_token'],\n // Deny-by-default egress: EVERY probed host must be listed here, plus the\n // chat host ('discord.com' for Discord, 'chat.googleapis.com' for a Google\n // Chat space webhook). Private, loopback and internal hosts are unreachable\n // by design \u2014 expose a public health endpoint guarded by the probe token.\n egressAllow: ['status.example.com', 'api.example.com', 'hooks.slack.com'],\n limits: { timeoutMs: 15000 }, // per outbound fetch; a check's own timeout_ms is \u2264 this\n },\n\n // DEAD-MAN'S SWITCH. Your own cron jobs call it when they finish:\n // POST /v1/fn/heartbeat {\"check\":\"nightly-backup\"}\n // with a key holding ONLY functions:invoke, kept in their CI secret store.\n heartbeat: {\n entry: './functions/heartbeat.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write'],\n egressAllow: [],\n },\n\n // STAFF ACTIONS: ack \xB7 resolve \xB7 post_update (outages.manage) and pause \xB7\n // resume \xB7 publish_component (checks.manage), each re-checked live against\n // the caller's org role.\n 'outage-action': {\n entry: './functions/outage-action.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write', 'orgs:read'],\n egressAllow: [],\n },\n\n // THE MORNING E-MAIL. 08:00 UTC \u2014 function crons are UTC; if you need local\n // time, enqueue it from a jobs schedule that carries a timezone instead.\n 'daily-report': {\n entry: './functions/daily-report.ts',\n trigger: { kind: 'cron', schedule: '0 8 * * *', overlap: 'skip' },\n scopes: ['cms:read', 'notifications:send'],\n egressAllow: [],\n },\n },\n\n // References only \u2014 values are stored once and never appear in this file.\n secrets: {\n slack_webhook_url: { feature: 'functions', description: 'Slack incoming webhook (or Discord / Google Chat space webhook) URL for pages' },\n probe_token: { feature: 'functions', description: 'shared token your public health endpoints require in x-probe-token' },\n oidc_client_secret: { feature: 'auth', description: 'OIDC client secret for your identity provider' },\n },\n\n // Seeds land as DRAFTS (a seed is a plain create). Publish the two components\n // once \u2014 `outage-action` `publish_component`, or the dashboard \u2014 and they\n // appear on the status page. The http checks land PAUSED (state 'paused',\n // enabled false) until their hosts are in the probe's egressAllow; turn each\n // on with `outage-action` `resume`, whose precondition is `state: 'paused'`.\n seed: {\n cms: [\n {\n collection: 'status_components',\n items: [\n { key: 'api', name: 'API', group: 'Core', status: 'operational', sort: 1, updated_at: '2026-01-01T00:00:00Z' },\n { key: 'website', name: 'Website', group: 'Core', status: 'operational', sort: 2, updated_at: '2026-01-01T00:00:00Z' },\n ],\n },\n {\n collection: 'checks',\n items: [\n {\n key: 'api-health', name: 'API health', kind: 'http', component: 'api', state: 'paused',\n url: 'https://api.example.com/health', expect_status: 200, send_probe_token: true,\n timeout_ms: 10000, degraded_ms: 2000, interval_min: 1, failures_to_down: 2,\n enabled: false, consecutive_failures: 0, next_due_at: '2026-01-01T00:00:00Z',\n },\n {\n key: 'website-home', name: 'Website', kind: 'keyword', component: 'website', state: 'paused',\n url: 'https://status.example.com/', expect_status: 200, expect_contains: '<title>',\n timeout_ms: 10000, degraded_ms: 3000, interval_min: 5, failures_to_down: 2,\n enabled: false, consecutive_failures: 0, next_due_at: '2026-01-01T00:00:00Z',\n },\n {\n key: 'nightly-backup', name: 'Nightly backup', kind: 'heartbeat', component: 'api', state: 'up',\n heartbeat_grace_min: 1500, interval_min: 15, failures_to_down: 1,\n enabled: true, consecutive_failures: 0, next_due_at: '2026-01-01T00:00:00Z',\n },\n ],\n },\n ],\n },\n});\n",
18508
+ "readme": '# Service monitor (ops)\n\nUptime and health checks for your team\'s **public endpoints**, **dead-man\'s-switch heartbeats** from your\nown scheduled jobs, outages that open and close **once per state crossing**, the on-call shift **paged from\ndata** in chat, e-mail and the in-app inbox, staff who acknowledge, resolve and post updates, and a\n**keyless public status page** that shows component status and only the updates you chose to publish.\n\nIt is four functions and six collections. There is no monitoring engine, no escalation engine and no\nconnector in it: a probe is a cron function behind a deny-by-default egress allowlist, an outage is a row\nwhose creation is guarded, a page is a read of who is on shift right now, and escalation is one more\nfilter in the same cron.\n\n```bash\nmkdir service-monitor && cd service-monitor && vxil init --template service-monitor # init scaffolds into the current directory\nvxil quickstart --env staging --no-push --invite <code> # or `vxil link <slug> --env staging`\nvxil projects workload <slug> staging # Free plan: functions deploy on staging/development\nprintf \'%s\' "$SLACK_URL" | vxil secrets set functions/slack_webhook_url\nprintf \'%s\' "$PROBE_TOKEN" | vxil secrets set functions/probe_token\nprintf \'%s\' "$OIDC_SECRET" | vxil secrets set auth/oidc_client_secret\nvxil push # collections, hooks, auth, orgs, the four functions\nvxil seed # 2 components, 2 http checks (paused), 1 heartbeat\n```\n\nThen, once:\n\n```bash\n# the two custom roles (rows, not config \u2014 `viewer` and `owner` are built in)\nvxil api POST /v1/orgs/roles --data \'{"role_key":"oncall","name":"On call","permissions":["outages.manage"]}\'\nvxil api POST /v1/orgs/roles --data \'{"role_key":"monitor-admin","name":"Monitor admin","permissions":["outages.manage","checks.manage"]}\'\n# the team org (its owner holds every permission), then each staff member with a role\n# \u2014 a member\'s user id exists after their first sign-in\nvxil api POST /v1/orgs --data \'{"slug":"platform","name":"Platform team","owner_user_id":"<your user id>"}\'\nvxil api POST /v1/orgs/<org_id>/members --data \'{"user_id":"<user id>","role":"oncall"}\'\n# who is on call, as data\nvxil api POST /v1/cms/items/oncall_shifts --data \'{"status":"published","data":{"user":"<user id>","role":"primary","starts_at":"2026-10-06T08:00:00Z","ends_at":"2026-10-13T08:00:00Z"}}\'\n```\n\nEdit `issuer`, `clientId` and `allowedDomains` in `vxil.config.ts` to your identity provider before you push.\n\n**Both probe secrets must be set before the probe can run**: a function whose declared secret has no value\nis refused at invoke (`409 secret_missing`), so the cron would fail every tick. No chat? Delete\n`\'secret:slack_webhook_url\'` from the probe\'s `secrets` (the probe then pages by e-mail and inbox only). No\ntoken-guarded endpoint yet? Set `probe_token` to any random value, or delete it the same way.\n\n> **Plan note.** Free deploys functions to `staging` and `development` projects, caps a project at 5\n> functions and 3 crons, and runs a cron every 15 minutes at the fastest. This blueprint uses 4 functions\n> and 2 crons, so it fits \u2014 change the probe\'s `schedule` to `*/15 * * * *` on Free (the config comment\n> says so), or the deploy answers `402 plan_limit`. Developer and above run it every minute.\n\n## The collections\n\n| Collection | What it holds | Who reads it |\n|---|---|---|\n| `checks` | what to watch: `key`, `kind` (`http` \xB7 `keyword` \xB7 `heartbeat`), `url`, expected status/text, timeout, `degraded_ms`, `interval_min`, `failures_to_down`, and the live state (`up` \xB7 `degraded` \xB7 `down` \xB7 `paused`) | staff |\n| `uptime_hourly` | one row per check per hour: `ok`, `total`, `latency_sum`, `latency_max` | staff, the daily report |\n| `outages` | one row per down-crossing: `open` \u2192 `acked` \u2192 `resolved`, who acked, duration, who was paged | staff |\n| `outage_updates` \u2B22 | the updates staff write; **only published ones are public** | everyone (published only) |\n| `status_components` \u2B22 | the components on your status page and their status | everyone (published only) |\n| `oncall_shifts` | who is on call, `primary` or `secondary`, from when to when | staff, the probe |\n\nThree Lane-A hooks guard the data in the same write: `outage_stage` (open \u2192 acked \u2192 resolved, resolved is\nfinal), `check_kind` and `check_url` (an http or keyword check needs an `https://` url), and `uptime_key`,\nwhich derives `point_key = check|bucket` server-side on a field declared `unique` \u2014 so one hour can never\nhave two rows for the same check.\n\n## Probes: a cron function with deny-by-default egress\n\n`probe` runs every minute. It reads the due checks \u2014 `enabled`, `next_due_at \u2264 now`, oldest first, at most\n`MAX_PER_TICK` (40) \u2014 and probes them in parallel:\n\n- **http** \u2014 `GET url`, with the check\'s `timeout_ms` (\u2264 15 s), `redirect: \'manual\'`, and the\n `x-probe-token` header when the check sets `send_probe_token`. Up when the status equals `expect_status`\n (200 by default).\n- **keyword** \u2014 the same, and the body must contain `expect_contains`.\n- **heartbeat** \u2014 no fetch at all (below).\n\nA **redirect is a failure** with reason `redirect`. The probe never follows one: a public host can redirect\nto a private one, so the platform\'s egress guard refuses to follow redirects, and the probe reports it\nrather than chasing it. Point a check at the final URL.\n\n**Every probed host must be on the probe\'s `egressAllow`.** A host that is not listed fails with\n`host not in egressAllow` \u2014 nothing leaves the function unless you named it. Add the chat host too\n(`hooks.slack.com`; `discord.com` for Discord; `chat.googleapis.com` for a Google Chat space webhook, which\ntakes `{ text }` alone \u2014 the function sends only that field to it).\n\n**What cannot be probed.** Private networks, loopback and internal hostnames are unreachable from a\nfunction by design. To watch an internal service, expose a small **public health endpoint** for it and\nguard it with the probe token:\n\n```text\nGET https://api.example.com/health x-probe-token: <probe_token>\n\u2192 200 when the service and its dependencies answer, 503 otherwise; 401 without the token\n```\n\nThe token is a per-function secret (`secret:probe_token`), so it never appears in your config or your\nrepository.\n\nThe seeded http checks land **paused** (`state: "paused"`, `enabled: false`) until their hosts are on the\nallowlist: edit the URLs, list the hosts, `vxil push`, then turn each one on with\n`{"op":"resume","check":"api-health"}` (and `website-home`) through `outage-action`. `resume` only accepts a\npaused check, sets it `up` and due now, and the next tick probes it.\n\n## Dead-man\'s-switch heartbeats\n\nSome failures make no request fail: a nightly backup that silently stopped running. A `heartbeat` check\ninverts the probe \u2014 **your job calls vxil** when it finishes, and silence is the alarm:\n\n```bash\n# the last step of the nightly backup job, wherever it runs\ncurl -s -X POST https://api.vxil.com/v1/fn/heartbeat \\\n -H "authorization: Bearer $VXIL_HEARTBEAT_KEY" -H \'content-type: application/json\' \\\n -d \'{"check":"nightly-backup"}\'\n```\n\n`VXIL_HEARTBEAT_KEY` is a **Server-class** key holding **only `functions:invoke`** (the `job-heartbeat` row in\n[the key table](#keys)), stored in that job\'s CI secret store. The\nfunction stamps `last_heartbeat_at` and makes the check due now; the probe marks the check down when the\nlast heartbeat is older than `heartbeat_grace_min` (no fetch, no egress). A check that has never beaten is\nnot judged: there is no `pending` state, so it keeps the state it has (`up` for a new check) and each probe\nrun only reschedules it, reporting it with `pending: true` in that run\'s result until the first heartbeat\narrives. The function answers `404` for an unknown check,\n`409` for a paused one or a non-heartbeat check, and `403` when it is called from a signed-in session:\nheartbeats are machine calls.\n\nA `functions:invoke` key can call any of the project\'s functions over HTTP, not only this one. Treat it\nlike any other credential, and keep `outage-action` safe on its own: it refuses a caller without a staff\nsession.\n\n## Once-per-crossing outages, paged from data\n\nThe state machine is small:\n\n- **up \u2192 down** after `failures_to_down` **consecutive** failures (2 by default). One failure is a blip.\n- **down \u2192 up** on the **first** success.\n- **degraded** when a successful answer took longer than `degraded_ms` \u2014 the component turns `degraded`\n and the chat gets one line, but no outage opens.\n\nEach tick first **claims** the check with one `If-Match` write that also moves `next_due_at` forward. A\nre-delivered or overlapping tick gets `409` there and stands down, so everything after the claim happens\nonce.\n\nOn the down-crossing the probe creates the outage under a **lock and a guard**:\n\n```json\nPOST /v1/cms/items/outages\n{ "data": { "check": "api-health", "state": "open", "severity": "major", \u2026 },\n "status": "published",\n "lock": "outage:api-health",\n "guard": { "filter": { "check": "api-health", "state": { "$in": ["open", "acked"] } }, "max": 1 } }\n```\n\nA check can therefore never have two open outages: a second create is `409 guard_failed`, which the probe\ntreats as *already open* and adopts. It then **pages the shift that covers now** \u2014 `oncall_shifts` with\n`starts_at \u2264 now < ends_at` and role `primary` \u2014 in chat once, and to each person by e-mail and in the\nin-app inbox. The page is claimed on the outage with an `if: { "paged": null }` precondition, so however\nthe tick is delivered, an outage pages at most once.\n\n**Escalation is not an engine.** In the same tick, the probe looks for outages still `open` (nobody acked)\n`ESCALATE_AFTER_MIN` (15) minutes after they opened, stamps `escalated_at` on each with a precondition, and\npages the `secondary`. Set the constant to `0` to turn it off.\n\nOn recovery the probe resolves the outage with an `if` on its state (`open` or `acked`), writes\n`duration_min`, and posts **recovered** once. If a staff member resolved it by hand while the check was\nstill failing, the probe does not reopen it until the check has recovered and failed again.\n\n## Keys\n\nMint the keys in the dashboard under **Keys \u2192 New key** (or with `vxil keys mint`, which needs a dashboard\nsession: `vxil login` first). Two of them live in the staff app, one in your jobs:\n\n| Key | Class | Scopes | Where it lives | Why |\n|---|---|---|---|---|\n| `web-auth` | Server | `auth:signin` | the staff app | signing in is how a session is obtained, so this key cannot require one |\n| `web-data` | Public / thin-client | `cms:read`, `functions:invoke` | the staff app | refused by the edge without a valid staff session |\n| `job-heartbeat` | Server | `functions:invoke` | your job\'s CI secret store, **never a browser** | heartbeats are machine calls with no session |\n\n`web-data` holds **no `cms:write`**: the staff app reads the collections directly, and every write goes\nthrough `outage-action`, which checks the person\'s permission. Because it is a thin-client key, every call\nit makes carries a signed-in session, and `heartbeat` refuses any call that carries one. **The staff app\ntherefore cannot send a heartbeat.**\n\n**Never ship a Server-class key with `functions:invoke` in a browser.** Anyone can copy a key out of a\npage. A Server-class key with no session can call `heartbeat` for any check key, as often as it likes, and\neach fake beat keeps that check `up`. A backup that stopped running would then never page anyone, which\ndefeats the heartbeat check. The heartbeat\'s only authentication is the key: the function accepts any caller\nholding `functions:invoke` *without* a session. Keep `job-heartbeat` in the job\'s secret store, give it no\nother scope, and rotate it like any other credential.\n\n`cms:read` on `web-data` lets **anyone who can sign in** read the staff collections, including check URLs,\noutages and the on-call rota. Signing in is fenced only by `allowedDomains`, so that means anyone at your\ncompany domain, not just members of the team org. Org roles gate the actions, not the reads. If the rota or\nthe internal URLs should be narrower than that, assign the identity-provider app to the on-call group only.\n\n## Staff actions\n\nYour staff app calls `outage-action` with the `web-data` key plus the signed-in staff member\'s session:\n\n```json\nPOST /v1/fn/outage-action\n{ "op": "ack", "outage_id": "\u2026" }\n{ "op": "resolve", "outage_id": "\u2026" }\n{ "op": "post_update", "outage_id": "\u2026", "kind": "identified", "body": "\u2026", "publish": true,\n "component_status": "partial_outage" }\n{ "op": "pause", "check": "api-health" }\n{ "op": "resume", "check": "api-health" }\n{ "op": "publish_component", "key": "api" }\n```\n\n`ack`, `resolve` and `post_update` need **`outages.manage`**; `pause`, `resume` and `publish_component` need\n**`checks.manage`**. The function re-checks the permission **live** on every call\n(`GET /v1/orgs/session-claims`), not from the role in the session, so a member you remove loses the buttons\nat once. Every state change is a conditional write and the `outage_stage` hook is the authority, so two\npeople pressing *Ack* at once produce one ack and one `409`.\n\n## The public status page\n\n`status_components` and `outage_updates` are `public: true`. Their **published** rows are readable with **no\nAPI key**, edge-cached, from any front end:\n\n```bash\ncurl "https://api.vxil.com/v1/cms/public/<TENANT_ID>/status_components?sort=sort"\ncurl "https://api.vxil.com/v1/cms/public/<TENANT_ID>/outage_updates?sort=-at&limit=20"\n```\n\n- **Only published rows are served.** A `post_update` with `publish: false` \u2014 and every `kind: "note"` \u2014\n is written as a **draft**: staff see it, the public never does. Notes are where internal detail belongs.\n- **Never put an internal URL, hostname or customer name in a published update.** The body is public the\n moment it is published.\n- The `author` of an update is role-gated (`readRoles`), so the public lane never serves your staff\'s ids.\n- Seeds land as drafts. Publish the two seeded components once:\n `{"op":"publish_component","key":"api"}` and `{"op":"publish_component","key":"website"}`.\n- The checks, their URLs, the outages and the shifts are **not** public. A component\'s status moves when\n the probe sees a crossing, or when staff publish an update with `component_status`.\n\n## Uptime counters\n\nOne `uptime_hourly` row per check per hour. The first sample of an hour creates the row (`point_key` is\nunique, so a racing create is a `409`, which turns into an increment on the existing row); every later\nsample is one atomic write:\n\n```json\nPATCH /v1/cms/items/uptime_hourly/<id>\n{ "$inc": { "ok": 1, "total": 1, "latency_sum": 182.4 } }\n```\n\n`$inc` is a single conditional statement \u2014 no read first, no lost update. `latency_max` is written only\nwhen a sample beats the hour\'s maximum.\n\n```\nuptime % = 100 \xD7 \u03A3 ok / \u03A3 total over the rows in the window\navg latency = \u03A3 latency_sum / \u03A3 ok (successful samples only)\n```\n\n`daily-report` (08:00 UTC) computes the first one for every check in **one server-side aggregate** \u2014\ngroup by `check`, sum `ok` and `total`, a 1-day window on `ts` \u2014 and e-mails it, worst first, with the\noutages opened that day, to whoever is on shift. Function crons run in UTC; for local time, enqueue the\nreport from a jobs schedule that carries a timezone.\n\n## Compose it\n\n- **alerts-to-slack** watches the *platform\'s own* failure events \u2014 this blueprint\'s functions failing,\n dead-lettered notifications, a quarantined function. Run both: one watches your services, the other\n watches the monitor.\n- **payments-heartbeat** is the same once-per-crossing pattern applied to silence from a payments\n integration\'s provider.\n- An **ops dashboard** in the same project can read `checks`, `outages` and `uptime_hourly` for its tiles \u2014\n the collection names here are specific to monitoring, so they do not clash.\n\n`vxil push` writes each feature\'s config as a whole, so to combine blueprints in one project, copy the\ncollections, hooks and functions of one into the other\'s `vxil.config.ts` and push that file.\n\n## Cadence and counts\n\n| | Free | Developer and above |\n|---|---|---|\n| probe cron | `*/15 * * * *` | `* * * * *` |\n| functions / crons used | 4 of 5 / 2 of 3 | 4 / 2 |\n| checks per tick | 40 (oldest due first) | 40 \u2014 raise `MAX_PER_TICK` with care: each check costs a probe and 2\u20133 writes |\n| finest interval | 15 minutes | 1 minute |\n\nA tick probes up to 40 checks; the rest wait one tick. With the default one-minute interval that is about\n40 checks per project; at `interval_min: 5` it is about 200.\n\n## What to learn from this\n\n- **Deny-by-default egress is a feature of a monitor.** The probe can reach exactly the hosts you listed \u2014\n and the public health endpoint plus a token is how an internal service joins the list safely.\n- **A crossing is a write that can only succeed once.** A lock + guard, an `If-Match` claim, and an `if`\n precondition turn an at-least-once cron into exactly one outage, one page and one "recovered".\n- **On-call is data.** Who to page is a filtered read of `oncall_shifts`, and escalation is the same cron\n asking one more question.\n- **Public by publishing.** A keyless status page needs no server: two public collections, and a draft\n that stays internal until someone chooses to publish it.\n',
18509
+ "functions": {
18510
+ "daily-report.ts": "// daily-report.ts \u2014 THE MORNING E-MAIL (a vxil function, cron trigger).\n//\n// Trigger: cron `0 8 * * *`. Function crons run in UTC; if your team wants it\n// at 08:00 local time, enqueue it from a jobs schedule that carries a timezone.\n//\n// ONE server-side aggregate over uptime_hourly \u2014 group by check, sum ok, sum\n// total, over the last day (the `ts` window) \u2014 gives every check's uptime:\n//\n// uptime % = 100 \xD7 \u03A3 ok / \u03A3 total (per check, last 24 h)\n//\n// plus the outages opened in the same window, e-mailed to whoever is on shift\n// now. A check with no samples in the window is simply absent from the table.\n\n// cron-walk: single-read \u2014 one bounded server-side aggregate (\u2264500 groups) and one bounded outage list are the whole job; nothing to page.\n\nimport type { CronFunctionEnvelope } from '@vxil/sdk';\n\nconst MAX_OUTAGES_LISTED = 50;\nconst MAX_LINES = 60; // checks listed in the e-mail body\n\ninterface Group { key: { check?: string }; ok?: number; total?: number }\ninterface Row<T> { item_id: string; data: T }\ninterface OutageData { check: string; state?: string; title?: string; opened_at?: string; duration_min?: number }\ninterface ShiftData { user: string; email?: string }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as CronFunctionEnvelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n const notify = env.scoped_jwts?.notifications;\n if (!cms || !notify) return Response.json({ error: 'missing cms or notifications scope' }, { status: 403 });\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n const now = new Date();\n const nowIso = now.toISOString();\n const since = new Date(now.getTime() - 86_400_000).toISOString();\n\n // 1. the aggregate: sum needs ok/total on n-slots (n1, n2); the 1-day window rides ts (t1)\n const agg = await fetch(`${base}/v1/cms/items/uptime_hourly/aggregate`, {\n method: 'POST', headers: H,\n body: JSON.stringify({\n aggregates: [\n { fn: 'sum', field: 'ok', as: 'ok' },\n { fn: 'sum', field: 'total', as: 'total' },\n ],\n groupBy: ['check'],\n window: { field: 'ts', sinceDays: 1 },\n limit: 500,\n }),\n });\n if (!agg.ok) return Response.json({ error: 'aggregate_failed', status: agg.status }, { status: 502 });\n const groups = ((await agg.json()) as { data?: { groups?: Group[] } }).data?.groups ?? [];\n\n // 2. the outages opened in the same window\n const of = encodeURIComponent(JSON.stringify({ opened_at: { $gte: since } }));\n const ol = await fetch(`${base}/v1/cms/items/outages?filter=${of}&sort=-opened_at&limit=${MAX_OUTAGES_LISTED}`, { headers: H });\n const outages = ol.ok ? ((await ol.json()) as { data?: { items?: Row<OutageData>[] } }).data?.items ?? [] : [];\n\n // 3. who gets it: everyone on shift right now\n const sf = encodeURIComponent(JSON.stringify({ starts_at: { $lte: nowIso }, ends_at: { $gt: nowIso } }));\n const sl = await fetch(`${base}/v1/cms/items/oncall_shifts?filter=${sf}&limit=20`, { headers: H });\n const shifts = sl.ok ? ((await sl.json()) as { data?: { items?: Row<ShiftData>[] } }).data?.items ?? [] : [];\n if (shifts.length === 0) return Response.json({ checks: groups.length, outages: outages.length, sent: 0, reason: 'nobody on shift' });\n\n const rows = groups\n .filter((g) => g.key.check && (g.total ?? 0) > 0)\n .map((g) => ({ check: g.key.check!, pct: (100 * (g.ok ?? 0)) / (g.total ?? 1), total: g.total ?? 0 }))\n .sort((a, b) => a.pct - b.pct); // worst first\n const lines = rows.slice(0, MAX_LINES).map((r) => `${r.check}: ${r.pct.toFixed(2)} % of ${r.total} checks`);\n if (rows.length > MAX_LINES) lines.push(`\u2026 and ${rows.length - MAX_LINES} more`);\n const outageLines = outages.map((o) => `${o.data.title ?? o.data.check} \u2014 ${o.data.state ?? 'open'}${o.data.duration_min !== undefined ? ` (${o.data.duration_min} min)` : ''}`);\n const paragraph = [\n `Uptime, last 24 h (worst first): ${lines.length ? lines.join(' \xB7 ') : 'no samples'}.`,\n `Outages opened: ${outageLines.length ? outageLines.join(' \xB7 ') : 'none'}.`,\n ].join(' ');\n\n // one report per person per day, however often this tick is delivered\n const day = nowIso.slice(0, 10);\n let sent = 0;\n for (const s of dedupe(shifts.map((r) => r.data))) {\n const res = await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: { authorization: `Bearer ${notify}`, 'content-type': 'application/json', 'idempotency-key': `daily-report:${day}:${s.user}` },\n body: JSON.stringify({\n user_id: s.user,\n ...(s.email ? { to_email: s.email } : {}),\n template: 'transactional',\n channel: 'email',\n data: { subject: `Service monitor \u2014 daily report ${day}`, paragraph },\n }),\n }).catch(() => null);\n if (res?.ok) sent++;\n }\n return Response.json({ checks: rows.length, outages: outages.length, sent });\n },\n};\n\nfunction dedupe(shifts: ShiftData[]): ShiftData[] {\n const seen = new Set<string>();\n return shifts.filter((s) => (seen.has(s.user) ? false : (seen.add(s.user), true)));\n}\n",
18511
+ "heartbeat.ts": "// heartbeat.ts \u2014 THE DEAD-MAN'S SWITCH (a vxil function, http trigger).\n//\n// Your own scheduled jobs (a backup, an export, a nightly sync \u2014 wherever they\n// run) call this when they FINISH:\n//\n// curl -s -X POST https://api.vxil.com/v1/fn/heartbeat \\\n// -H \"authorization: Bearer $VXIL_HEARTBEAT_KEY\" -H 'content-type: application/json' \\\n// -d '{\"check\":\"nightly-backup\"}'\n//\n// with a SERVER-class key holding ONLY `functions:invoke`, kept in that job's CI\n// secret store and never shipped to a browser: the key is this endpoint's only\n// authentication, so whoever holds it can keep a dead job's check `up`. The\n// staff app's thin-client key always carries a session, which is refused here. It stamps `last_heartbeat_at` on the check and makes the check due now,\n// so the next probe tick (\u2264 1 minute) judges it \u2014 a check that was down\n// recovers on its first beat. The PROBE is what notices silence: a check whose\n// last heartbeat is older than its grace goes down there, with no fetch.\n//\n// 404 \u2014 no such check \xB7 409 \u2014 not a heartbeat check, or paused/disabled\n// 403 \u2014 called from a signed-in session: heartbeats are machine calls\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\ntype Env = HttpFunctionEnvelope<{ check?: string }>;\ninterface CheckData { key: string; kind?: string; state?: string; enabled?: boolean }\ninterface Row { item_id: string; data: CheckData }\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n if (!cms) return fail(403, 'missing_scope', 'the function has no cms token');\n if (env.end_user) return fail(403, 'machine_only', 'a heartbeat comes from a job with a functions:invoke key, not from a signed-in session');\n const key = env.payload?.check;\n if (typeof key !== 'string' || !/^[A-Za-z0-9._:-]{1,120}$/.test(key)) {\n return fail(422, 'bad_check', 'send {\"check\":\"<check key>\"}');\n }\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n\n const filter = encodeURIComponent(JSON.stringify({ key }));\n const found = await fetch(`${base}/v1/cms/items/checks?filter=${filter}&limit=1`, { headers: H });\n if (!found.ok) return fail(502, 'lookup_failed', `checks read ${found.status}`);\n const row = ((await found.json()) as { data?: { items?: Row[] } }).data?.items?.[0];\n if (!row) return fail(404, 'unknown_check', `no check '${key}'`);\n if (row.data.kind !== 'heartbeat') return fail(409, 'not_a_heartbeat', `'${key}' is a ${row.data.kind} check`);\n if (row.data.state === 'paused' || row.data.enabled === false) return fail(409, 'paused', `'${key}' is paused`);\n\n const now = new Date().toISOString();\n const res = await fetch(`${base}/v1/cms/items/checks/${row.item_id}`, {\n method: 'PATCH',\n headers: H,\n // `if`: a check paused between the read and this write stays paused\n body: JSON.stringify({ data: { last_heartbeat_at: now, next_due_at: now }, if: { state: { $ne: 'paused' } } }),\n });\n if (res.status === 409) return fail(409, 'paused', `'${key}' is paused`);\n if (!res.ok) return fail(502, 'write_failed', `checks write ${res.status}`);\n return Response.json({ check: key, beat_at: now });\n },\n};\n\nconst fail = (status: number, code: string, message: string) => Response.json({ error: { code, message } }, { status });\n",
18512
+ "outage-action.ts": "// outage-action.ts \u2014 WHAT STAFF DO DURING AN OUTAGE (a vxil function, http trigger).\n//\n// Called from your staff app with a browser key that holds `functions:invoke`\n// AND the signed-in staff member's session (the X-Vxil-End-User header), so\n// `env.end_user` is the VERIFIED caller:\n//\n// POST /v1/fn/outage-action\n// { \"op\": \"ack\", \"outage_id\": \"\u2026\" } outages.manage\n// { \"op\": \"resolve\", \"outage_id\": \"\u2026\" } outages.manage\n// { \"op\": \"post_update\", \"outage_id\": \"\u2026\", \"kind\": \"identified\",\n// \"body\": \"\u2026\", \"publish\": true, \"component_status\": \"partial_outage\" } outages.manage\n// { \"op\": \"pause\", \"check\": \"api-health\" } checks.manage\n// { \"op\": \"resume\", \"check\": \"api-health\" } checks.manage\n// { \"op\": \"publish_component\", \"key\": \"api\" } checks.manage\n//\n// Every op re-checks the permission LIVE against the caller's org role (the\n// role in the session is a snapshot; a member you just removed must lose the\n// buttons now, not at their next refresh). `owner` holds every permission.\n//\n// Every state change is a conditional write (`if` on the current state), and\n// the cms lifecycle hook is the authority on legal transitions \u2014 so two staff\n// pressing \"Ack\" at once produce one ack and one clean 409.\n//\n// post_update with publish:false (or kind 'note') writes a DRAFT: visible to\n// staff, never served on the public status page.\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\ntype Op = 'ack' | 'resolve' | 'post_update' | 'pause' | 'resume' | 'publish_component';\ninterface Payload {\n op?: Op; outage_id?: string; check?: string; key?: string;\n kind?: string; body?: string; publish?: boolean; component_status?: string;\n}\ntype Env = HttpFunctionEnvelope<Payload>;\ninterface Row<T> { item_id: string; version?: number; status?: string; data: T }\ninterface OutageData { check: string; state?: string; opened_at?: string; title?: string }\ninterface CheckData { key: string; component?: string; state?: string }\n\nconst NEEDS: Record<Op, 'outages.manage' | 'checks.manage'> = {\n ack: 'outages.manage', resolve: 'outages.manage', post_update: 'outages.manage',\n pause: 'checks.manage', resume: 'checks.manage', publish_component: 'checks.manage',\n};\nconst UPDATE_KINDS = ['investigating', 'identified', 'monitoring', 'resolved', 'note'];\nconst COMPONENT_STATUSES = ['operational', 'degraded', 'partial_outage', 'major_outage'];\nconst BODY_MAX = 2000;\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n const orgs = env.scoped_jwts?.orgs;\n if (!cms || !orgs) return fail(403, 'missing_scope', 'the function needs its cms and orgs tokens');\n // Staff actions are PEOPLE's actions: a server key with no session has no\n // one to check a role for, so it is refused rather than trusted.\n const user = env.end_user?.id;\n if (!user) return fail(403, 'staff_session_required', 'call this with a signed-in staff session');\n const p = env.payload ?? {};\n const op = p.op;\n if (!op || !(op in NEEDS)) return fail(422, 'bad_op', `op must be one of ${Object.keys(NEEDS).join(', ')}`);\n\n // \u2500\u2500 the permission, checked live \u2500\u2500\n const claims = await fetch(`${base}/v1/orgs/session-claims?user_id=${encodeURIComponent(user)}`, {\n headers: { authorization: `Bearer ${orgs}` },\n });\n if (!claims.ok) return fail(502, 'role_lookup_failed', `orgs ${claims.status}`);\n const perms = ((await claims.json()) as { data?: { perms?: string[] } }).data?.perms ?? [];\n const need = NEEDS[op];\n if (!perms.includes('*') && !perms.includes(need)) return fail(403, 'forbidden', `this needs the ${need} permission`);\n\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n const now = new Date().toISOString();\n\n // \u2500\u2500 outages.manage \u2500\u2500\n if (op === 'ack' || op === 'resolve' || op === 'post_update') {\n if (!p.outage_id) return fail(422, 'bad_outage', 'send outage_id');\n const got = await fetch(`${base}/v1/cms/items/outages/${encodeURIComponent(p.outage_id)}`, { headers: H });\n if (got.status === 404) return fail(404, 'unknown_outage', 'no such outage');\n if (!got.ok) return fail(502, 'lookup_failed', `outages read ${got.status}`);\n const outage = ((await got.json()) as { data?: Row<OutageData> }).data;\n if (!outage) return fail(404, 'unknown_outage', 'no such outage');\n\n if (op === 'ack') {\n const r = await fetch(`${base}/v1/cms/items/outages/${outage.item_id}`, {\n method: 'PATCH', headers: H,\n body: JSON.stringify({ data: { state: 'acked', acked_by: user }, if: { state: 'open' } }),\n });\n if (r.status === 409) return fail(409, 'not_open', `the outage is ${outage.data.state}`);\n if (!r.ok) return relay(r);\n return Response.json({ outage_id: outage.item_id, state: 'acked', acked_by: user });\n }\n\n if (op === 'resolve') {\n const opened = Date.parse(outage.data.opened_at ?? '');\n const durationMin = Number.isFinite(opened) ? Math.round(((Date.parse(now) - opened) / 60_000) * 10) / 10 : undefined;\n const r = await fetch(`${base}/v1/cms/items/outages/${outage.item_id}`, {\n method: 'PATCH', headers: H,\n body: JSON.stringify({\n data: { state: 'resolved', resolved_at: now, ...(durationMin !== undefined ? { duration_min: durationMin } : {}) },\n if: { state: { $in: ['open', 'acked'] } },\n }),\n });\n if (r.status === 409) return fail(409, 'already_resolved', 'the outage is already resolved');\n if (!r.ok) return relay(r);\n return Response.json({ outage_id: outage.item_id, state: 'resolved', duration_min: durationMin ?? null });\n }\n\n // post_update\n const kind = p.kind ?? 'note';\n if (!UPDATE_KINDS.includes(kind)) return fail(422, 'bad_kind', `kind must be one of ${UPDATE_KINDS.join(', ')}`);\n const body = typeof p.body === 'string' ? p.body.trim() : '';\n if (!body || body.length > BODY_MAX) return fail(422, 'bad_body', `body is 1\u2013${BODY_MAX} characters`);\n if (p.component_status !== undefined && !COMPONENT_STATUSES.includes(p.component_status)) {\n return fail(422, 'bad_component_status', `component_status must be one of ${COMPONENT_STATUSES.join(', ')}`);\n }\n // A note is internal by definition: it is never published.\n const publish = p.publish === true && kind !== 'note';\n const component = await componentOfCheck(base, H, outage.data.check);\n const created = await fetch(`${base}/v1/cms/items/outage_updates`, {\n method: 'POST', headers: H,\n body: JSON.stringify({\n data: { outage: outage.item_id, kind, author: user, at: now, body, ...(component ? { component } : {}) },\n status: publish ? 'published' : 'draft',\n }),\n });\n if (!created.ok) return relay(created);\n const itemId = ((await created.json()) as { data?: { item_id?: string } }).data?.item_id ?? null;\n\n // Optionally move the PUBLIC component status with a published update.\n let componentStatus: string | null = null;\n if (publish && p.component_status && component) {\n const filter = encodeURIComponent(JSON.stringify({ key: component }));\n const list = await fetch(`${base}/v1/cms/items/status_components?filter=${filter}&limit=1`, { headers: H });\n const comp = list.ok ? ((await list.json()) as { data?: { items?: Row<unknown>[] } }).data?.items?.[0] : undefined;\n if (comp) {\n const r = await fetch(`${base}/v1/cms/items/status_components/${comp.item_id}`, {\n method: 'PATCH', headers: H,\n body: JSON.stringify({ data: { status: p.component_status, updated_at: now } }),\n });\n if (r.ok) componentStatus = p.component_status;\n }\n }\n return Response.json({\n update_id: itemId, published: publish, ...(componentStatus ? { component, component_status: componentStatus } : {}),\n }, { status: 201 });\n }\n\n // \u2500\u2500 checks.manage \u2500\u2500\n if (op === 'publish_component') {\n if (!p.key) return fail(422, 'bad_key', 'send key (a status_components key)');\n const filter = encodeURIComponent(JSON.stringify({ key: p.key }));\n const list = await fetch(`${base}/v1/cms/items/status_components?filter=${filter}&limit=1`, { headers: H });\n if (!list.ok) return fail(502, 'lookup_failed', `status_components read ${list.status}`);\n const comp = ((await list.json()) as { data?: { items?: Row<unknown>[] } }).data?.items?.[0];\n if (!comp) return fail(404, 'unknown_component', `no component '${p.key}'`);\n if (comp.status === 'published') return Response.json({ key: p.key, published: true, already: true });\n const r = await fetch(`${base}/v1/cms/items/status_components/${comp.item_id}/publish`, { method: 'POST', headers: H });\n if (!r.ok) return relay(r);\n return Response.json({ key: p.key, published: true });\n }\n\n if (!p.check) return fail(422, 'bad_check', 'send check (a checks key)');\n const filter = encodeURIComponent(JSON.stringify({ key: p.check }));\n const list = await fetch(`${base}/v1/cms/items/checks?filter=${filter}&limit=1`, { headers: H });\n if (!list.ok) return fail(502, 'lookup_failed', `checks read ${list.status}`);\n const check = ((await list.json()) as { data?: { items?: Row<CheckData>[] } }).data?.items?.[0];\n if (!check) return fail(404, 'unknown_check', `no check '${p.check}'`);\n\n const pausing = op === 'pause';\n const r = await fetch(`${base}/v1/cms/items/checks/${check.item_id}`, {\n method: 'PATCH', headers: H,\n body: JSON.stringify(pausing\n ? { data: { state: 'paused', enabled: false }, if: { state: { $ne: 'paused' } } }\n // resume: due now, counting from zero; the next tick judges it afresh\n : { data: { state: 'up', enabled: true, consecutive_failures: 0, next_due_at: now }, if: { state: 'paused' } }),\n });\n if (r.status === 409) return fail(409, pausing ? 'already_paused' : 'not_paused', `the check is ${check.data.state}`);\n if (!r.ok) return relay(r);\n return Response.json({ check: p.check, state: pausing ? 'paused' : 'up' });\n },\n};\n\nasync function componentOfCheck(base: string, H: Record<string, string>, checkKey: string): Promise<string | null> {\n const filter = encodeURIComponent(JSON.stringify({ key: checkKey }));\n const res = await fetch(`${base}/v1/cms/items/checks?filter=${filter}&limit=1`, { headers: H });\n if (!res.ok) return null;\n return ((await res.json()) as { data?: { items?: Row<CheckData>[] } }).data?.items?.[0]?.data.component ?? null;\n}\n\n/** Pass a refused cms write through (a hook rejection is a 422 with its message). */\nasync function relay(r: Response): Promise<Response> {\n const detail = (await r.text().catch(() => '')).slice(0, 400);\n return Response.json({ error: { code: 'write_refused', status: r.status, detail } }, { status: r.status >= 500 ? 502 : r.status });\n}\n\nconst fail = (status: number, code: string, message: string) => Response.json({ error: { code, message } }, { status });\n",
18513
+ "probe.ts": "// probe.ts \u2014 UPTIME + HEALTH CHECKS, ONCE-PER-CROSSING OUTAGES (a vxil function).\n//\n// Trigger: cron `* * * * *` (Free plan: `*/15 * * * *`). Each tick:\n// 1. reads the DUE checks \u2014 enabled, next_due_at \u2264 now, oldest first, at most\n// MAX_PER_TICK \u2014 and probes them in parallel:\n// \u2022 http / keyword: GET the https url with the check's own timeout,\n// `redirect: 'manual'` (a redirect is a FAILURE, reason 'redirect'),\n// the `x-probe-token` header when the check opts in, then the expected\n// status and (keyword) the expected text;\n// \u2022 heartbeat: no fetch \u2014 down when the last heartbeat is older than\n// the check's grace;\n// 2. CLAIMS each check for this tick with one If-Match PATCH that also moves\n// next_due_at forward (so the row leaves the due filter) \u2014 a re-delivered\n// or overlapping tick gets 409 and stands down, so nothing below runs twice;\n// 3. counts the sample into ONE uptime_hourly row per check per hour (create\n// on the first sample of the hour, atomic $inc after);\n// 4. acts on a STATE CROSSING:\n// up \u2192 down after failures_to_down consecutive failures: open the outage\n// under lock `outage:<check>` + a guard (\u22641 open/acked outage\n// per check \u2014 a second create is a 409, treated as \"already\n// open\"), set the component status, page whoever is on shift\n// NOW (chat once; e-mail + inbox per person);\n// down \u2192 up on the first success: resolve the outage with its duration,\n// post \"recovered\" once;\n// up \u2194 degraded latency above degraded_ms: component status + one chat line;\n// 5. ESCALATES: an outage still `open` (nobody acked) ESCALATE_AFTER_MIN after\n// it opened pages the secondary \u2014 one more filter in the same cron, not an\n// escalation engine.\n//\n// It never throws: a check that fails to probe is a failed sample, a check\n// whose bookkeeping fails is reported in the result, and the tick answers 500\n// only when it could not read its own checks.\n\n// cron-walk: drains-filter \u2014 each probed check is PATCHed with a later next_due_at, so it leaves the due filter until its next interval.\n\nimport type { CronFunctionEnvelope, HttpFunctionEnvelope } from '@vxil/sdk';\n\nconst MAX_PER_TICK = 40; // checks probed per tick (oldest-due first; the rest wait one tick)\nconst DEFAULT_TIMEOUT_MS = 10_000;\nconst MAX_TIMEOUT_MS = 15_000; // = the function's limits.timeoutMs\nconst ESCALATE_AFTER_MIN = 15; // page the secondary when nobody acked; 0 = never\nconst MAX_ESCALATIONS_PER_TICK = 10;\nconst ERROR_MAX = 300;\nconst KEYWORD_SCAN_MAX = 1_000_000; // characters of the body searched for expect_contains\n// Where the page's button points: your staff app's outage screen.\nconst ADMIN_URL = 'https://status-admin.example.com/outages/';\n\ntype State = 'up' | 'degraded' | 'down' | 'paused';\ntype Envelope = CronFunctionEnvelope | HttpFunctionEnvelope<Record<string, unknown>>;\n\ninterface CheckData {\n key: string; kind: 'http' | 'keyword' | 'heartbeat'; component?: string; state?: State;\n consecutive_failures?: number; last_latency_ms?: number; next_due_at?: string; last_checked_at?: string;\n name?: string; url?: string; expect_status?: number; expect_contains?: string; send_probe_token?: boolean;\n timeout_ms?: number; degraded_ms?: number; interval_min?: number; failures_to_down?: number;\n enabled?: boolean; heartbeat_grace_min?: number; last_heartbeat_at?: string;\n hour_bucket?: string; hour_row?: string; hour_max?: number; open_outage?: string; last_error?: string;\n}\ninterface OutageData {\n check: string; state?: 'open' | 'acked' | 'resolved'; severity?: string; opened_at?: string;\n title?: string; paged?: unknown; escalated_at?: string;\n}\ninterface ShiftData { user: string; email?: string; role?: 'primary' | 'secondary'; starts_at?: string; ends_at?: string }\ninterface ComponentData { key: string; name?: string; status?: string }\ninterface Row<T> { item_id: string; version?: number; data: T }\n\n/** One probe result. `pending` = a heartbeat check that has never beaten. */\ninterface Sample { ok: boolean; latencyMs?: number; reason?: string; pending?: boolean }\n\ninterface Ctx {\n cms: Cms;\n base: string;\n notify?: string | undefined; // the notifications scoped token\n chatUrl?: string | undefined;\n probeToken?: string | undefined;\n now: Date;\n nowIso: string;\n shifts?: Promise<Row<ShiftData>[]>;\n}\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Envelope;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cmsJwt = env.scoped_jwts?.cms;\n if (!cmsJwt) return Response.json({ error: 'missing cms scope' }, { status: 403 });\n const now = new Date();\n const ctx: Ctx = {\n cms: new Cms(base, cmsJwt),\n base,\n notify: env.scoped_jwts?.notifications,\n chatUrl: env.secrets?.slack_webhook_url || undefined,\n probeToken: env.secrets?.probe_token || undefined,\n now,\n nowIso: now.toISOString(),\n };\n\n // 1. the due checks \u2014 the work queue IS the filter\n let due: Row<CheckData>[];\n try {\n due = await ctx.cms.list<CheckData>(\n 'checks', { enabled: true, next_due_at: { $lte: ctx.nowIso } }, 'next_due_at', MAX_PER_TICK,\n );\n } catch (e) {\n return Response.json({ error: 'checks_unreadable', detail: errText(e) }, { status: 500 });\n }\n\n // 2\u20134. probe in parallel; each check's bookkeeping is independent\n const results = await Promise.all(due.map((row) => runCheck(ctx, row).catch((e) => ({\n check: row.data.key, error: errText(e),\n }))));\n\n // 5. escalation \u2014 one more filter in the same cron\n const escalated = await escalate(ctx).catch(() => 0);\n\n return Response.json({ probed: due.length, escalated, results });\n },\n};\n\n// \u2500\u2500 one check, one tick \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\nasync function runCheck(ctx: Ctx, row: Row<CheckData>): Promise<Record<string, unknown>> {\n const d = row.data;\n const sample = d.kind === 'heartbeat' ? heartbeatSample(d, ctx.now) : await probeHttp(d, ctx.probeToken);\n const nextDue = new Date(ctx.now.getTime() + Math.max(1, d.interval_min ?? 1) * 60_000).toISOString();\n\n if (sample.pending) {\n // A heartbeat check that has never beaten: nothing to judge yet. Just reschedule.\n const claimed = await ctx.cms.patch('checks', row.item_id, { next_due_at: nextDue, last_checked_at: ctx.nowIso }, { ifMatch: row.version });\n return { check: d.key, pending: true, claimed };\n }\n\n const prev: State = d.state === 'paused' || !d.state ? 'up' : d.state;\n const failures = sample.ok ? 0 : (d.consecutive_failures ?? 0) + 1;\n const next = decide(prev, sample, failures, d.failures_to_down ?? 2, d.degraded_ms);\n const bucket = ctx.nowIso.slice(0, 13); // 'YYYY-MM-DDTHH' (UTC)\n const newHour = d.hour_bucket !== bucket || !d.hour_row;\n const prevMax = newHour ? 0 : (d.hour_max ?? 0);\n const okLatency = sample.ok ? sample.latencyMs : undefined;\n\n // CLAIM: If-Match on the version we read. The PATCH also moves next_due_at\n // forward, so this row leaves the due filter \u2014 a re-delivered tick reads it\n // no more, and an overlapping one gets 409 here and stands down.\n const claimed = await ctx.cms.patch('checks', row.item_id, {\n state: next,\n consecutive_failures: failures,\n last_checked_at: ctx.nowIso,\n next_due_at: nextDue,\n last_error: sample.ok ? null : (sample.reason ?? 'failed').slice(0, ERROR_MAX),\n ...(sample.latencyMs !== undefined ? { last_latency_ms: sample.latencyMs } : {}),\n hour_max: Math.max(prevMax, okLatency ?? 0),\n }, { ifMatch: row.version });\n if (!claimed) return { check: d.key, skipped: 'claimed by another tick' };\n\n await countUptime(ctx, row, sample, bucket, newHour, prevMax);\n\n // \u2500\u2500 state crossings \u2500\u2500\n const actions: string[] = [];\n if (next === 'down' && !d.open_outage) {\n // the up\u2192down crossing \u2014 or a crossing whose outage was never recorded\n // (the guard makes this idempotent: an already-open outage is a 409)\n actions.push(await openOutage(ctx, row, sample.reason ?? 'failed'));\n }\n if (prev === 'down' && next !== 'down') {\n actions.push(await resolveOutage(ctx, row));\n }\n if (prev !== 'down' && next !== 'down' && prev !== next) {\n const label = displayName(d);\n await post(ctx.chatUrl, next === 'degraded'\n ? `\u{1F7E0} DEGRADED: ${label} answered in ${Math.round(sample.latencyMs ?? 0)} ms (threshold ${d.degraded_ms} ms)`\n : `\u{1F7E2} ${label} is fast again (${Math.round(sample.latencyMs ?? 0)} ms)`);\n actions.push(next === 'degraded' ? 'degraded' : 'latency-recovered');\n }\n if (prev !== next && d.component) await updateComponent(ctx, d.component);\n\n return { check: d.key, ok: sample.ok, state: next, prev, ...(sample.reason ? { reason: sample.reason } : {}), ...(actions.length ? { actions } : {}) };\n}\n\n/** The state machine. Down only after N consecutive failures; up on the first success. */\nfunction decide(prev: State, s: Sample, failures: number, failuresToDown: number, degradedMs?: number): State {\n if (!s.ok) return failures >= failuresToDown ? 'down' : prev === 'down' ? 'down' : prev;\n if (degradedMs && (s.latencyMs ?? 0) > degradedMs) return 'degraded';\n return 'up';\n}\n\n// \u2500\u2500 the probes \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\nfunction heartbeatSample(d: CheckData, now: Date): Sample {\n const last = Date.parse(d.last_heartbeat_at ?? '');\n if (!Number.isFinite(last)) return { ok: true, pending: true };\n const graceMin = d.heartbeat_grace_min ?? 1440;\n const ageMin = (now.getTime() - last) / 60_000;\n return ageMin > graceMin\n ? { ok: false, reason: `no heartbeat for ${fmtMin(ageMin)} (grace ${fmtMin(graceMin)})` }\n : { ok: true };\n}\n\nasync function probeHttp(d: CheckData, probeToken?: string): Promise<Sample> {\n if (!d.url || !d.url.startsWith('https://')) return { ok: false, reason: 'no https url' };\n const timeout = Math.min(MAX_TIMEOUT_MS, Math.max(1000, d.timeout_ms ?? DEFAULT_TIMEOUT_MS));\n const headers: Record<string, string> = { 'user-agent': 'vxil-service-monitor' };\n if (d.send_probe_token && probeToken) headers['x-probe-token'] = probeToken;\n const t0 = Date.now();\n let res: Response;\n try {\n res = await fetch(d.url, { method: 'GET', headers, redirect: 'manual', signal: AbortSignal.timeout(timeout) });\n } catch (e) {\n const name = (e as { name?: string })?.name;\n return { ok: false, latencyMs: Date.now() - t0, reason: name === 'TimeoutError' || name === 'AbortError' ? `timeout after ${timeout} ms` : 'connection failed' };\n }\n const latencyMs = Date.now() - t0;\n\n // The egress guard answers for the network when the request never reached the\n // host: a host missing from egressAllow, a redirect it refused to follow, its\n // own timeout. It marks those answers with `x-vxil-egress`.\n const egress = res.headers.get('x-vxil-egress');\n if (egress) {\n const body = (await res.json().catch(() => ({}))) as { reason?: string };\n if (String(body.reason ?? '').startsWith('redirect')) return { ok: false, latencyMs, reason: 'redirect' };\n if (egress === 'blocked') return { ok: false, latencyMs, reason: 'host not in egressAllow' };\n if (egress === 'timeout') return { ok: false, latencyMs, reason: `timeout after ${timeout} ms` };\n return { ok: false, latencyMs, reason: 'connection failed' };\n }\n if (res.status >= 300 && res.status < 400) {\n await res.body?.cancel();\n return { ok: false, latencyMs, reason: 'redirect' };\n }\n const expect = d.expect_status ?? 200;\n if (res.status !== expect) {\n await res.body?.cancel();\n return { ok: false, latencyMs, reason: `status ${res.status} (expected ${expect})` };\n }\n if (d.kind === 'keyword' && d.expect_contains) {\n const text = (await res.text().catch(() => '')).slice(0, KEYWORD_SCAN_MAX);\n if (!text.includes(d.expect_contains)) return { ok: false, latencyMs, reason: 'expected text not found' };\n } else {\n await res.body?.cancel();\n }\n return { ok: true, latencyMs };\n}\n\n// \u2500\u2500 uptime counters: one row per check per hour \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\nasync function countUptime(ctx: Ctx, row: Row<CheckData>, s: Sample, bucket: string, newHour: boolean, prevMax: number): Promise<void> {\n const d = row.data;\n const lat = s.ok ? (s.latencyMs ?? 0) : 0;\n const inc = { ok: s.ok ? 1 : 0, total: 1, latency_sum: lat };\n\n if (!newHour && d.hour_row) {\n const r = await ctx.cms.inc('uptime_hourly', d.hour_row, inc);\n if (r === 'ok') {\n // the slowest sample this hour \u2014 written only when it is a new maximum\n if (s.ok && lat > prevMax) await ctx.cms.patch('uptime_hourly', d.hour_row, { latency_max: lat });\n return;\n }\n if (r !== 'missing') return; // anything but a deleted row: leave the counter alone\n }\n\n // The first sample of the hour: create the row. point_key (check|bucket) is\n // derived server-side and unique, so a racing create is a 409 \u2014 then the row\n // exists, and this sample is an increment on it.\n let id = await ctx.cms.create('uptime_hourly', {\n check: d.key, bucket, ts: `${bucket}:00:00.000Z`, ok: inc.ok, total: 1, latency_sum: lat, latency_max: lat,\n });\n if (!id) {\n const existing = (await ctx.cms.list<{ point_key: string }>('uptime_hourly', { point_key: `${d.key}|${bucket}` }, undefined, 1))[0];\n if (!existing) return;\n id = existing.item_id;\n await ctx.cms.inc('uptime_hourly', id, inc);\n }\n await ctx.cms.patch('checks', row.item_id, { hour_bucket: bucket, hour_row: id });\n}\n\n// \u2500\u2500 outages: once per crossing \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\nasync function openOutage(ctx: Ctx, row: Row<CheckData>, reason: string): Promise<string> {\n const d = row.data;\n // lock + guard: under the per-check lock, refuse the create when an open or\n // acked outage for this check already exists. A re-delivered tick, an\n // overlapping one, or a manual double-open all collapse into ONE row.\n let outageId = await ctx.cms.create('outages', {\n check: d.key, state: 'open', severity: 'major', opened_at: ctx.nowIso,\n title: `${displayName(d)} is down`, cause: reason.slice(0, ERROR_MAX),\n }, { lock: `outage:${d.key}`, guard: { filter: { check: d.key, state: { $in: ['open', 'acked'] } }, max: 1 } });\n let paged: unknown = null;\n if (!outageId) {\n // 409 guard_failed \u21D2 it is already open: adopt that one\n const open = (await ctx.cms.list<OutageData>('outages', { check: d.key, state: { $in: ['open', 'acked'] } }, '-opened_at', 1))[0];\n if (!open) return 'open-failed';\n outageId = open.item_id;\n paged = open.data.paged ?? null;\n }\n await ctx.cms.patch('checks', row.item_id, { open_outage: outageId });\n if (paged) return 'already-open';\n const who = await page(ctx, outageId, d, reason, 'primary');\n return who === null ? 'opened (already paged)' : `opened, paged ${who}`;\n}\n\nasync function resolveOutage(ctx: Ctx, row: Row<CheckData>): Promise<string> {\n const d = row.data;\n let outage: Row<OutageData> | null = d.open_outage ? await ctx.cms.get<OutageData>('outages', d.open_outage) : null;\n if (!outage) outage = (await ctx.cms.list<OutageData>('outages', { check: d.key, state: { $in: ['open', 'acked'] } }, '-opened_at', 1))[0] ?? null;\n let result = 'nothing-open';\n if (outage && (outage.data.state === 'open' || outage.data.state === 'acked')) {\n const opened = Date.parse(outage.data.opened_at ?? '');\n const durationMin = Number.isFinite(opened) ? Math.round(((ctx.now.getTime() - opened) / 60_000) * 10) / 10 : undefined;\n // the `if` precondition: exactly one writer resolves it (a staff member who\n // resolved it by hand already, or a concurrent tick, makes this a 409)\n const won = await ctx.cms.patch('outages', outage.item_id, {\n state: 'resolved', resolved_at: ctx.nowIso, ...(durationMin !== undefined ? { duration_min: durationMin } : {}),\n }, { if: { state: { $in: ['open', 'acked'] } } });\n if (won) {\n await post(ctx.chatUrl, `\u{1F7E2} RECOVERED: ${displayName(d)} is up again after ${durationMin !== undefined ? fmtMin(durationMin) : 'an outage'}`);\n result = 'resolved';\n } else {\n result = 'already-resolved';\n }\n }\n await ctx.cms.patch('checks', row.item_id, { open_outage: null });\n return result;\n}\n\n/** Page the shift that covers NOW. The claim on `paged` (an `if` precondition\n * that it is still empty) makes the page at-most-once per outage. Returns the\n * people paged, or null when someone else already paged. */\nasync function page(ctx: Ctx, outageId: string, d: CheckData, reason: string, tier: 'primary' | 'secondary'): Promise<string | null> {\n const shifts = await currentShifts(ctx);\n const wanted = shifts.filter((s) => (s.data.role ?? 'primary') === tier);\n const to = (wanted.length ? wanted : shifts).map((s) => s.data);\n if (tier === 'primary') {\n const claimed = await ctx.cms.patch('outages', outageId, {\n paged: { at: ctx.nowIso, to: to.map((s) => s.user), tier },\n }, { if: { paged: null } });\n if (!claimed) return null;\n }\n const names = to.map((s) => s.email ?? s.user).join(', ') || 'nobody is on shift';\n const head = tier === 'primary' ? '\u{1F534} DOWN' : `\u23EB NOT ACKED after ${ESCALATE_AFTER_MIN} min`;\n await post(ctx.chatUrl, `${head}: ${displayName(d)} \u2014 ${reason}. Paging: ${names}. ${ADMIN_URL}${outageId}`);\n for (const s of to) await sendPage(ctx, outageId, d, reason, s, tier);\n return names;\n}\n\nasync function sendPage(ctx: Ctx, outageId: string, d: CheckData, reason: string, s: ShiftData, tier: string): Promise<boolean> {\n if (!ctx.notify) return false;\n const res = await fetch(`${ctx.base}/v1/notifications/send`, {\n method: 'POST',\n headers: {\n authorization: `Bearer ${ctx.notify}`,\n 'content-type': 'application/json',\n // one page per (outage, person, tier) however often this runs\n 'idempotency-key': `page:${outageId}:${s.user}:${tier}`,\n },\n body: JSON.stringify({\n user_id: s.user,\n ...(s.email ? { to_email: s.email } : {}),\n template: 'transactional',\n channel: 'both', // e-mail + the in-app inbox\n data: {\n subject: `[${tier === 'primary' ? 'DOWN' : 'ESCALATED'}] ${displayName(d)}`,\n paragraph: `${displayName(d)} is down since ${ctx.nowIso}: ${reason}. Acknowledge it so the team knows someone is on it.`,\n cta_label: 'Open the outage',\n cta_url: `${ADMIN_URL}${outageId}`,\n },\n }),\n }).catch(() => null);\n return Boolean(res?.ok);\n}\n\n/** Nobody acked: page the secondary once. `escalated_at` takes the outage out\n * of this filter, so each outage escalates at most once. */\nasync function escalate(ctx: Ctx): Promise<number> {\n if (ESCALATE_AFTER_MIN <= 0) return 0;\n const before = new Date(ctx.now.getTime() - ESCALATE_AFTER_MIN * 60_000).toISOString();\n const stale = await ctx.cms.list<OutageData>(\n 'outages', { state: 'open', opened_at: { $lt: before }, escalated_at: null }, 'opened_at', MAX_ESCALATIONS_PER_TICK,\n );\n let n = 0;\n for (const o of stale) {\n const won = await ctx.cms.patch('outages', o.item_id, { escalated_at: ctx.nowIso }, { if: { escalated_at: null, state: 'open' } });\n if (!won) continue;\n const check = (await ctx.cms.list<CheckData>('checks', { key: o.data.check }, undefined, 1))[0]?.data ?? { key: o.data.check, kind: 'http' as const };\n await page(ctx, o.item_id, check, o.data.title ?? 'still down', 'secondary');\n n++;\n }\n return n;\n}\n\nfunction currentShifts(ctx: Ctx): Promise<Row<ShiftData>[]> {\n ctx.shifts ??= ctx.cms\n .list<ShiftData>('oncall_shifts', { starts_at: { $lte: ctx.nowIso }, ends_at: { $gt: ctx.nowIso } }, 'starts_at', 20)\n .catch(() => []);\n return ctx.shifts;\n}\n\n// \u2500\u2500 the public status of a component, from all of its checks \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\nasync function updateComponent(ctx: Ctx, key: string): Promise<void> {\n const checks = await ctx.cms.list<CheckData>('checks', { component: key, enabled: true }, undefined, 50);\n const states = checks.map((c) => c.data.state ?? 'up').filter((s) => s !== 'paused');\n const down = states.filter((s) => s === 'down').length;\n const status = states.length > 0 && down === states.length ? 'major_outage'\n : down > 0 ? 'partial_outage'\n : states.includes('degraded') ? 'degraded'\n : 'operational';\n const comp = (await ctx.cms.list<ComponentData>('status_components', { key }, undefined, 1))[0];\n if (!comp || comp.data.status === status) return;\n await ctx.cms.patch('status_components', comp.item_id, { status, updated_at: ctx.nowIso });\n}\n\n// \u2500\u2500 the cms client (REST envelope: { data: { items: [{ item_id, version, data }] } }) \u2500\u2500\n\nclass Cms {\n constructor(private base: string, private jwt: string) {}\n private h(extra: Record<string, string> = {}) {\n return { authorization: `Bearer ${this.jwt}`, 'content-type': 'application/json', ...extra };\n }\n async list<T>(coll: string, filter: Record<string, unknown>, sort: string | undefined, limit: number): Promise<Row<T>[]> {\n const q = `filter=${encodeURIComponent(JSON.stringify(filter))}${sort ? `&sort=${sort}` : ''}&limit=${limit}`;\n const res = await fetch(`${this.base}/v1/cms/items/${coll}?${q}`, { headers: this.h() });\n if (!res.ok) throw new Error(`list ${coll} ${res.status}`);\n return ((await res.json()) as { data?: { items?: Row<T>[] } }).data?.items ?? [];\n }\n async get<T>(coll: string, id: string): Promise<Row<T> | null> {\n const res = await fetch(`${this.base}/v1/cms/items/${coll}/${id}`, { headers: this.h() });\n if (!res.ok) return null;\n return ((await res.json()) as { data?: Row<T> }).data ?? null;\n }\n /** \u2192 the new item_id, or null on a 409 (a unique key or a guard already holds) */\n async create(coll: string, data: Record<string, unknown>, opts: { lock?: string; guard?: unknown } = {}): Promise<string | null> {\n const res = await fetch(`${this.base}/v1/cms/items/${coll}`, {\n method: 'POST', headers: this.h(), body: JSON.stringify({ data, status: 'published', ...opts }),\n });\n if (!res.ok) return null;\n return ((await res.json()) as { data?: { item_id?: string } }).data?.item_id ?? null;\n }\n /** \u2192 false on 409 (version / precondition) or any refusal */\n async patch(coll: string, id: string, data: Record<string, unknown>, opts: { ifMatch?: number | undefined; if?: unknown } = {}): Promise<boolean> {\n const res = await fetch(`${this.base}/v1/cms/items/${coll}/${id}`, {\n method: 'PATCH',\n headers: this.h(opts.ifMatch !== undefined ? { 'if-match': String(opts.ifMatch) } : {}),\n body: JSON.stringify({ data, ...(opts.if !== undefined ? { if: opts.if } : {}) }),\n });\n return res.ok;\n }\n /** atomic increment \u2014 one conditional statement, no read first */\n async inc(coll: string, id: string, by: Record<string, number>): Promise<'ok' | 'missing' | 'refused'> {\n const res = await fetch(`${this.base}/v1/cms/items/${coll}/${id}`, {\n method: 'PATCH', headers: this.h(), body: JSON.stringify({ $inc: by }),\n });\n return res.ok ? 'ok' : res.status === 404 ? 'missing' : 'refused';\n }\n}\n\n// \u2500\u2500 chat: `text` is Slack's field, `content` is Discord's \u2014 send both, EXCEPT to\n// a Google Chat space webhook (chat.googleapis.com), which takes `{ text }`\n// alone (an unknown field can be rejected there).\nasync function post(url: string | undefined, text: string): Promise<{ ok: boolean; status: number }> {\n if (!url) return { ok: false, status: 0 };\n let host = '';\n try { host = new URL(url).hostname; } catch { /* unparseable \u2192 the generic body below */ }\n const body = host === 'chat.googleapis.com' ? { text } : { text, content: text };\n const res = await fetch(url, {\n method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body),\n }).catch(() => null);\n return { ok: Boolean(res?.ok), status: res?.status ?? 0 };\n}\n\nconst displayName = (d: CheckData) => d.name || d.key;\nconst errText = (e: unknown) => String((e as Error)?.message ?? e).slice(0, ERROR_MAX);\nfunction fmtMin(min: number): string {\n if (min < 120) return `${Math.round(min)} min`;\n if (min < 2880) return `${Math.round((min / 60) * 10) / 10} h`;\n return `${Math.round((min / 1440) * 10) / 10} d`;\n}\n"
18514
+ }
18515
+ },
18397
18516
  {
18398
18517
  "id": "job-runner",
18399
18518
  "title": "Job Runner (queues \xB7 cron \xB7 generation \xB7 credits)",
@@ -18443,11 +18562,11 @@ export default defineConfig({
18443
18562
  "vxil_jobs_key"
18444
18563
  ],
18445
18564
  "configSrc": "import { defineConfig } from '@vxil/config';\n\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n// \"Render Farm\" \u2014 long renders and transcodes (minutes, ffmpeg, a headless\n// browser) run on a runtime YOU rent: Trigger.dev, Modal, a container on your\n// own cloud account. vxil keeps the four things that must not be lost while\n// that runtime works: the credits held for the render, the deadline, the\n// signed completion callback, and the status row the app watches.\n//\n// request-render (your function, end-user mode)\n// \u2192 creates-or-finds the user's `renders` row (unique render_key)\n// \u2192 POST /v1/jobs/generation in WEBHOOK mode: your render endpoint,\n// credits HELD, a deadline, the status mirrored onto the row\n// vxil's generation lane\n// \u2192 POSTs your render endpoint { render_id, user_id, composition, props,\n// payload, callback_url } with your bearer token\n// your runtime (README: Trigger.dev, or your own containers)\n// \u2192 acks within 20 s, renders, POSTs { status: 'processing', \u2026 } and then\n// { status: 'completed', output_url, duration_s, \u2026 } to callback_url\n// vxil\n// \u2192 completed: credits COMMIT, every callback key lands on the row\n// \u2192 failed / no answer by the deadline: credits REFUNDED, row says failed\n// \u2192 job.generation.completed | failed \u2192 notify-ready tells the owner\n// redrive-pending (cron, every minute)\n// \u2192 starts renders that waited at the concurrency cap (429 \u2014 the app was\n// told `queued: true`), same key; tells the owner if one never starts\n//\n// No container tier and no workflow engine inside vxil: the runtime is yours,\n// the bookkeeping is vxil's. \"Credits\" are usage units on the deterministic\n// `mock` payments integration \u2014 not money; vxil is never in the flow of funds.\n// \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\nexport default defineConfig({\n env: 'staging',\n\n features: {\n jobs: {\n enabled: true,\n generation: {\n // how many renders may be in flight at once for this project\n maxConcurrent: 20,\n // a render that never calls back fails (and refunds) after 30 minutes\u2026\n defaultTimeoutMs: 1_800_000,\n // \u2026and no render may ask for more than the platform ceiling, one hour\n maxTimeoutMs: 3_600_000,\n // one render never holds more than 50 credits\n maxReserveCredits: 50,\n // all of this project's in-flight renders together hold at most 5,000\n maxOutstandingReserveCredits: 5_000,\n },\n },\n\n payments: {\n enabled: true,\n provider: 'mock',\n defaults: { currency: 'usd' },\n ledger: {\n productMap: {\n render_pack_100: { creditType: 'render_credits', amount: 100, period: 'once' },\n },\n // no subscription tiers in this blueprint \u2014 credits come from packs\n tierMap: {},\n // a render that fails, times out or is cancelled gives its credits back\n autoRefundOnJobFailure: true,\n },\n },\n\n cms: {\n // a render row is live the moment it is written\n draftPublish: false,\n // in end-user mode a signed-in user sees only the renders they own\n strictEndUserScope: true,\n // Lane-A hook (guide ch. 7): render_key IS owner + ':' + request_key,\n // server-enforced, so one user's request_key can never collide with \u2014\n // or block \u2014 another user's.\n hooks: {\n render_key_shape: {\n collection: 'renders',\n event: 'beforeWrite',\n kind: 'validate',\n expr: \"item.render_key == concat(item.owner, ':', item.request_key)\",\n message: \"render_key must be owner + ':' + request_key\",\n },\n },\n },\n\n notifications: { provider: 'mock', fromEmail: 'renders@render-farm.example' },\n functions: { enabled: true },\n\n // the README's keyless-container coordinator uploads each finished output\n // into this project's files (`output_file` below). The per-object ceiling\n // the upload-url call pre-checks is `maxObjectBytes` \u2014 100 MB by default,\n // which fits the coordinator's buffered upload (\"tens of MB\"); raise it\n // (up to 5 GB) for long or high-bitrate renders. Not using the coordinator?\n // Remove this and the `output_file` field.\n files: { enabled: true },\n },\n\n cms: {\n collections: {\n renders: {\n singular: 'render',\n ownerField: 'owner',\n fields: {\n // THE DEDUPE ANCHOR: a double tap or a retried request is a 409 that\n // request-render reads back \u2014 and the same key is the generation\n // run's idempotency_key, so the render endpoint is asked once.\n render_key: { type: 'string', required: true, unique: true, indexSlot: 's1' },\n request_key: { type: 'string', required: true },\n owner: { type: 'string', indexSlot: 's2' },\n // which composition / preset your runtime renders (your vocabulary)\n composition: { type: 'string', required: true, indexSlot: 's3' },\n // written by vxil's status mirror: pending \u2192 processing \u2192 completed | failed\n status: { type: 'string', indexSlot: 's4' },\n credits: { type: 'int', indexSlot: 'n1' },\n created_at: { type: 'datetime', indexSlot: 't1' },\n props: { type: 'json' },\n run_id: { type: 'text' },\n // \u2500\u2500 keys your runtime sends back. EVERY key of the completion body is\n // written onto this row, so each one must be a declared field (a\n // write naming an unknown field is refused: vxil then writes the\n // status word alone and the run reports `mirror_error` with the\n // keys it dropped \u2014 declare the field so its value lands too).\n output_url: { type: 'text' },\n // the files-feature object id, when a coordinator uploads the output\n // into this project's files (README \"Keys stay out of the container\")\n output_file: { type: 'file' },\n duration_s: { type: 'float' },\n // progress keys: a `processing` ping carries them onto the row\n // (request-render's status_mirror.progress_fields \u2014 progress 0..100,\n // a float: the run keeps 2 decimals,\n // stage \u2264 64 chars, message \u2264 200), and the completion body writes\n // them too (progress: 100, stage: 'done').\n progress: { type: 'float', validation: { min: 0, max: 100 } },\n stage: { type: 'string' },\n message: { type: 'text' },\n // written by notify-ready from job.generation.failed (and by\n // redrive-pending when a render never got a run)\n error: { type: 'text' },\n // written by redrive-pending: how often it tried to start this render,\n // and when it last did. The re-driver's read needs no new index \u2014\n // `status` (s4) and `created_at` (t1) are slots; `run_id: null` is a\n // residual test over the rows they pick.\n redrive_attempts: { type: 'int' },\n redriven_at: { type: 'datetime' },\n },\n },\n },\n },\n\n functions: {\n // Starts ONE render for the signed-in user. Invoke it in END-USER mode\n // (with the user's session): the held credits are forced onto that user,\n // and the row is theirs.\n 'request-render': {\n entry: './functions/request-render.ts',\n trigger: { kind: 'http' },\n scopes: ['cms:read', 'cms:write', 'jobs:write'],\n // render_url: your render endpoint (https). render_token: the bearer\n // token that endpoint checks. Both ride the generation run; vxil's\n // generation lane \u2014 not this function \u2014 calls the endpoint.\n secrets: ['secret:render_url', 'secret:render_token'],\n egressAllow: [],\n signature: {\n input: { composition: 'string', request_key: 'string', props: 'json?' },\n output:\n '{ item_id: string; run_id: string; credits: number }'\n + ' | { duplicate: true; request_key: string; item_id: string; run_id: string | null; status: string }',\n },\n },\n\n // THE BACKLOG RE-DRIVER. At the generation cap request-render answers 429\n // and the row waits `pending` with no run; every minute this starts the\n // oldest such rows (\u2265 30 s old, 20 per tick) with the SAME idempotency key,\n // stops at the first 429 (or at a refusal that means the setup is wrong),\n // and fails \u2014 and tells the owner of \u2014 a row that waited over an hour.\n // overlap 'skip': a slow tick is never doubled by the next one.\n // Free plan: a function cron may fire at most every 15 minutes \u2014 use\n // '*/15 * * * *' there (README \"The backlog\").\n 'redrive-pending': {\n entry: './functions/redrive-pending.ts',\n trigger: { kind: 'cron', schedule: '* * * * *', overlap: 'skip' },\n scopes: ['cms:read', 'cms:write', 'notifications:send'],\n // vxil_jobs_key: an API key of this backend holding ONLY jobs:write \u2014 a\n // cron tick has no signed-in user to hold credits for, so the start is\n // made as your trusted server (README \"The backlog\")\n secrets: ['secret:vxil_jobs_key', 'secret:render_url', 'secret:render_token'],\n egressAllow: [],\n },\n\n // job.generation.completed | failed \u2192 write the failure cause onto the row\n // and tell the owner. A non-2xx is retried; the notification's\n // Idempotency-Key (one per run) makes a redelivery send nothing twice.\n 'notify-ready': {\n entry: './functions/notify-ready.ts',\n trigger: { kind: 'webhook', source: 'job.generation.', retry: { maxAttempts: 3 } },\n scopes: ['cms:read', 'cms:write', 'notifications:send'],\n egressAllow: [],\n },\n },\n\n secrets: {\n render_url: {\n feature: 'functions',\n description: 'your render endpoint \u2014 the https URL vxil POSTs each render to (a Trigger.dev relay, or your own container endpoint)',\n },\n render_token: {\n feature: 'functions',\n description: 'a long random token your render endpoint checks on the Authorization header (Bearer \u2026)',\n },\n vxil_jobs_key: {\n feature: 'functions',\n description: 'an API key of this backend holding ONLY jobs:write \u2014 redrive-pending starts backlogged renders with it (vxil keys mint --name render-redrive --scopes jobs:write). jobs:write also lets it cancel or replay any run, enqueue jobs and manage schedules and flow rules: a server key, kept only here',\n },\n },\n});\n",
18446
- "readme": "# Render Farm \u2014 long renders on a runtime you rent, with vxil holding the credits, the deadline and the callback\n\n```bash\nvxil init my-renders --template render-farm\ncd my-renders\nprintf '%s' \"$RENDER_URL\" | vxil secrets set functions/render_url # your render endpoint (https)\nprintf '%s' \"$RENDER_TOKEN\" | vxil secrets set functions/render_token # a long random token it checks\n# the backlog re-driver's key: an API key of this backend holding ONLY jobs:write\nvxil keys mint --name render-redrive --scopes jobs:write --json | jq -r .api_key | vxil secrets set functions/vxil_jobs_key\nvxil push\n```\n\n> **Plan note.** The functions deploy on the Free plan when the project's workload is `staging` or\n> `development` (`vxil projects workload <slug> development`, or create it with\n> `vxil projects create <slug> --workload development`). On a Free `production` project, `vxil push` stops before it writes anything, naming the plan and the ways out: change the workload or upgrade to Developer, or run `vxil push --skip-functions` to apply the collections and config without the functions.\n> On the Free plan a function cron may also fire at most every 15 minutes, so change `redrive-pending`'s\n> schedule to `'*/15 * * * *'` there. The blueprint is written for Developer and up, where it runs every\n> minute. Everything else works the same; a backlog just drains more slowly.\n\nA video render, a transcode, a headless-browser capture: minutes of CPU, ffmpeg or Chromium. That does\nnot fit in a vxil function (a delivered trigger gets about a minute), and vxil will not grow a container\ntier or a workflow engine to run it. So the work runs on **a runtime you rent** \u2014 Trigger.dev, Modal,\na container on your own cloud account \u2014 and vxil keeps the four things that must survive while it\nruns:\n\n| vxil holds | so that |\n|---|---|\n| **the credits** reserved for the render | a failed, abandoned or cancelled render gives them back, and a user can never start more than they can pay for |\n| **the deadline** | a render your runtime never reports on fails and refunds after 30 minutes (at most one hour) |\n| **the signed completion callback** | your runtime needs no vxil key: the URL it is handed is the credential for that one render |\n| **the status row** | the app reads (or subscribes to) one `renders` row: `pending \u2192 processing \u2192 completed | failed`, plus everything your runtime sent back |\n\n**What this blueprint teaches that the others do not:** the hand-off to **your own** long-running\nruntime through a **webhook-mode generation run** \u2014 the contract your endpoint and your worker must\nkeep, and two complete runtime options below. (`fal-media` shows the same lane against a vendor queue\nAPI; `job-runner` shows a provider call vxil polls.)\n\n## What you get\n\n- **`renders`** \u2014 one row per render, owned by the user who asked for it (`strictEndUserScope`: a\n signed-in user reads only their own). `render_key` is the owner + `:` + the client's `request_key`\n (the composition is enforced by a `beforeWrite` hook) and is **unique**, so a double tap or a retried\n request finds the first row instead of starting a second render. Your runtime's answer lands on the\n row: `output_url`, `duration_s`, `progress`, `stage`, `message`.\n- **`request-render`** (http function, end-user mode) \u2014 creates-or-finds the row, then starts ONE\n generation run: your endpoint (`render_url`), your token on its `Authorization` header, a status\n mirror onto the row, `reserve_credits` for the render (5 `render_credits`), a 30-minute deadline, and\n the `render_key` as the run's `idempotency_key` \u2014 so a re-driven start gets the same run back. The\n credits it holds are the function's fixed price, never a value read from the row.\n- **`redrive-pending`** (cron function, every minute, `overlap: 'skip'`) \u2014 drains the backlog. When\n the project already has `generation.maxConcurrent` renders in flight, `request-render` answers `429`\n (with `queued: true`) and the row waits `pending` with no run. This function starts those rows,\n oldest first, with the **same** idempotency key, and tells the owner when one never starts\n ([the backlog](#the-backlog-maxconcurrent-429-and-the-re-driver)).\n- **`notify-ready`** (webhook function on `job.generation.`) \u2014 writes the failure cause onto the row (a\n platform class such as `GenerationExpired` gets a short human hint after it) and sends the owner one\n message per run (`Idempotency-Key: render-ready:<run_id>`). A render refused for too few credits\n keeps `insufficient_credits`; when `request-render` refused it, the caller already got the `402`\n and nothing is sent, and when the re-driver started it (the user last heard \"queued\"), the owner\n is told once (`Idempotency-Key: render-not-started:<item_id>`, the re-driver's own key).\n- **credits** \u2014 a payments integration on the `mock` provider (no provider account needed to try it).\n \"Credits\" are usage units you meter, not money.\n\nThe row is a **view** for the app; the run and the ledger are the truth. If your app's client key\ncarries `cms:write`, a signed-in user can edit their own `renders` row (say, set `status` to\n`completed`) \u2014 that changes nothing they are charged or given. Anything that grants something on\ncompletion should read the run (`GET /v1/jobs/runs/{run_id}`) or react to `job.generation.completed`,\nas `notify-ready` does \u2014 or give the client key only `cms:read`.\n\n## The contract your runtime keeps\n\nWhatever runs the render, these are the only things it has to do:\n\n| Step | What arrives / what to send |\n|---|---|\n| **Start** | vxil `POST`s your `render_url` with `Authorization: Bearer <render_token>` and JSON `{ render_id, user_id, composition, props, payload: { generation_id, correlation_id, deadline_at }, callback_url }` (`user_id` is the render's owner). Check the token, **queue the work and answer `2xx` within 20 seconds** \u2014 never render inline. `408` / `429` / `5xx` / a timeout is retried with backoff; any other `4xx` ends the run and refunds the credits. A start can arrive more than once (a lost answer is retried), so key the work by `render_id`. |\n| **Progress** (optional) | `POST callback_url` with `{\"status\": \"processing\", \"progress\": 40, \"stage\": \"encoding\"}`. A processing ping moves the row to `processing` and writes its `progress` (0\u2013100) / `stage` (\u2264 64 chars) / `message` (\u2264 200 chars) onto the row \u2014 `request-render` asks for that with `status_mirror.progress_fields` \u2014 so the app shows live progress by watching the row. Other keys on a ping are not written to the row (the run keeps the latest report, `GET /v1/jobs/runs/{run_id}` \u2192 `progress`). Keep pings to one every few seconds at most. |\n| **Done** | `POST callback_url` with `{\"status\": \"completed\", \"output_url\": \"https://\u2026\", \"duration_s\": 31.2, \"progress\": 100, \"stage\": \"done\"}` \u2014 or, when the output is uploaded into this project's files, `{\"status\": \"completed\", \"output_file\": \"obj_\u2026\", \"duration_s\": 31.2, \"progress\": 100, \"stage\": \"done\"}` ([below](#keys-stay-out-of-the-container-the-recommended-shape)). At most 256 KiB, and **every key a declared field of `renders`** (add a field before you send a new key; [test it](#a-contract-test-for-your-runtime)). The credits are committed and every key is written onto the row. |\n| **Failed** | `POST callback_url` with `{\"status\": \"failed\", \"error\": \"render_failed\", \"hint\": \"ffmpeg exited 1: \u2026\"}`. The credits are refunded; `error` (a short code) and `hint` reach `notify-ready` as `error_class` / `error_hint` and land on the row's `error`. |\n| **Never answers** | the run fails at the deadline (`GenerationExpired`), the credits are refunded, and a callback after that changes nothing. |\n| **The deadline** | `payload.deadline_at` (an ISO time) is when vxil stops waiting. Check it before each attempt starts: a render that cannot finish by then should post `failed` and stop. A `completed` posted after it is answered with the settled `failed` state \u2014 the output exists, but the user was refunded and the row says failed. |\n\n`callback_url` needs no other credential \u2014 and nothing else should see it. A repeated `completed` or\n`failed` post is answered with the settled state and changes nothing, so your worker can safely retry\nits own callback on a network error. The output bytes stay where your runtime wrote them (your bucket,\nyour CDN) and vxil stores the keys, not the file, unless a coordinator uploads the output into this\nproject's files ([next section](#keys-stay-out-of-the-container-the-recommended-shape)).\n\nEach start request also carries an `X-Vxil-Jobs-Signature` header (verifiable with your project's\njobs signing secret, `GET /v1/jobs/signing-secret`) and `x-vxil-run-id`. This blueprint uses the\nbearer token because it is one string comparison in any language.\n\n## Keys stay out of the container (the recommended shape)\n\nThe render container is the part of your system that runs the most third-party code (ffmpeg,\nChromium, fonts and media from the user's props), so give it **no vxil key at all**. This is the\nshape the blueprint recommends, whatever runs the container:\n\n```\nvxil \u2500\u2500start\u2500\u2500\u25B6 coordinator \u2500\u2500launch\u2500\u2500\u25B6 container\n (render_url) \u2502 progress / failed \u2500\u2500\u2500\u2500\u2500\u2500\u25B6 callback_url (keyless)\n \u25B2 output bytes \u2500\u2500\u2500\u2500\u2500\u2518\n \u2502\n \u2514\u2500 mints the upload URL with ITS files:write key, PUTs the bytes,\n completes the object, POSTs \"completed\" + output_file \u2500\u2500\u25B6 callback_url\n```\n\n- **The files feature is on.** The blueprint's config enables it (`files: { enabled: true }`) and\n declares `output_file` as a `file` field; without it every upload-url call is refused and the\n render waits out its deadline. Its per-object ceiling, `maxObjectBytes`, is 100 MB by default,\n enough for the buffered coordinator below; raise it for long or high-bitrate renders.\n- **The coordinator** is your `render_url`: a small endpoint in your own account \u2014 an edge worker\n with a per-render lock, or a route on any thin server. It is the only piece that holds a vxil key,\n and that key holds only **`files:write`** (`vxil keys mint --name render-uploads --scopes files:write`).\n- **The container** gets the job, the keyless `callback_url` (for `processing` pings and for\n `failed`), an `output_url` on the coordinator and an **output ticket** for it: a token signed for\n that one render and useless after its deadline, sent on the `Authorization` header (never in the\n URL, where access logs would keep it).\n- **The upload happens once the size is known.** The container POSTs the finished file to its\n ticket. The coordinator reads it, mints the files upload URL for exactly that size\n (the quota pre-check uses it), PUTs the bytes, completes the object, and only then posts\n `completed` with the object id as `output_file`. A crash anywhere before that leaves the render\n open, and it refunds at the deadline like any other.\n\n```ts\n// coordinator.ts \u2014 your render_url. A standard fetch handler (an edge worker, or a Node 18+ adapter).\n// Env: RENDER_TOKEN (= the vxil secret render_token), TICKET_SECRET (a long random string),\n// VXIL_FILES_KEY (an API key holding ONLY files:write), VXIL_BASE (https://api.vxil.com).\ntype Start = {\n render_id: string; user_id: string; composition: string; props?: Record<string, unknown>;\n payload: { generation_id: string; deadline_at: string }; callback_url: string;\n};\ntype Ticket = { render_id: string; user_id: string; callback_url: string; exp: number };\ntype Env = { RENDER_TOKEN: string; TICKET_SECRET: string; VXIL_FILES_KEY: string; VXIL_BASE: string };\n\nconst enc = new TextEncoder();\nconst b64u = (b: ArrayBuffer | Uint8Array) =>\n btoa(String.fromCharCode(...new Uint8Array(b))).replace(/\\+/g, '-').replace(/\\//g, '_').replace(/=+$/, '');\nconst unb64u = (s: string) => Uint8Array.from(atob(s.replace(/-/g, '+').replace(/_/g, '/')), (c) => c.charCodeAt(0));\nconst hmacKey = (secret: string) =>\n crypto.subtle.importKey('raw', enc.encode(secret), { name: 'HMAC', hash: 'SHA-256' }, false, ['sign', 'verify']);\nasync function sealTicket(env: Env, t: Ticket): Promise<string> {\n const body = b64u(enc.encode(JSON.stringify(t)));\n return `${body}.${b64u(await crypto.subtle.sign('HMAC', await hmacKey(env.TICKET_SECRET), enc.encode(body)))}`;\n}\nasync function openTicket(env: Env, raw: string): Promise<Ticket | null> {\n const [body, sig] = raw.split('.');\n if (!body || !sig) return null;\n try {\n // crypto.subtle.verify compares in constant time (a `!==` on the signature would not)\n if (!(await crypto.subtle.verify('HMAC', await hmacKey(env.TICKET_SECRET), unb64u(sig), enc.encode(body)))) return null;\n const t = JSON.parse(new TextDecoder().decode(unb64u(body))) as Ticket;\n return t.exp > Date.now() ? t : null;\n } catch {\n return null; // not base64url / not JSON\n }\n}\n/** The output's file extension, from the Content-Type the container sends. */\nconst EXT: Record<string, string> = {\n 'video/mp4': 'mp4', 'video/webm': 'webm', 'image/gif': 'gif', 'image/png': 'png', 'image/jpeg': 'jpg', 'application/pdf': 'pdf',\n};\nconst vxil = (env: Env, path: string, body?: unknown) => fetch(`${env.VXIL_BASE}${path}`, {\n method: 'POST',\n headers: { authorization: `Bearer ${env.VXIL_FILES_KEY}`, 'content-type': 'application/json' },\n ...(body ? { body: JSON.stringify(body) } : {}),\n});\n\nexport default {\n async fetch(req: Request, env: Env): Promise<Response> {\n const url = new URL(req.url);\n\n // 1. the start: check the token, launch the container, answer inside 20 s\n if (req.method === 'POST' && url.pathname === '/start') {\n if (req.headers.get('authorization') !== `Bearer ${env.RENDER_TOKEN}`) return new Response('unauthorized', { status: 401 });\n const s = (await req.json()) as Start;\n // a run started before request-render sent user_id (an older copy of this\n // blueprint): a server-mode upload must name its user, so refuse the start \u2014\n // a 4xx ends that run and refunds it, instead of a 422 at upload time\n if (!s.user_id) return new Response('start body has no user_id: redeploy request-render', { status: 400 });\n const ticket = await sealTicket(env, {\n render_id: s.render_id, user_id: s.user_id, callback_url: s.callback_url,\n exp: Date.parse(s.payload.deadline_at), // useless once vxil stops waiting\n });\n await launchContainer({ // YOUR container platform's API, keyed by\n render_id: s.render_id, // render_id: a start vxil re-sends launches once\n composition: s.composition, props: s.props ?? {},\n deadline_at: s.payload.deadline_at,\n callback_url: s.callback_url, // for processing pings and `failed`\n output_url: `${url.origin}/output`, // POST the file here, with\n output_ticket: ticket, // Authorization: Bearer <output_ticket>\n });\n return Response.json({ accepted: true }, { status: 202 });\n }\n\n // 2. the output: the container POSTs the finished file here, once\n if (req.method === 'POST' && url.pathname === '/output') {\n const t = await openTicket(env, (req.headers.get('authorization') ?? '').replace(/^Bearer /, ''));\n if (!t) return new Response('bad or expired ticket', { status: 403 });\n const contentType = (req.headers.get('content-type') ?? '').split(';')[0]!.trim().toLowerCase();\n const ext = EXT[contentType];\n if (!ext) return new Response(`send the output's Content-Type (one of: ${Object.keys(EXT).join(', ')})`, { status: 415 });\n // buffered: fine for outputs of tens of MB \u2014 see \"Very large outputs\" below\n const bytes = await req.arrayBuffer();\n const size = bytes.byteLength;\n if (size === 0) return new Response('empty output', { status: 400 });\n\n const minted = await vxil(env, '/v1/files/upload-url', {\n user_id: t.user_id, filename: `${t.render_id}.${ext}`, content_type: contentType, size_bytes: size,\n });\n if (!minted.ok) return new Response(`upload-url ${minted.status}`, { status: 502 }); // the container sends the file again\n const { object_id, upload_url } = ((await minted.json()) as { data: { object_id: string; upload_url: string } }).data;\n const put = await fetch(upload_url, { method: 'PUT', headers: { 'content-type': contentType }, body: bytes });\n if (!put.ok) return new Response(`upload ${put.status}`, { status: 502 });\n const done = await vxil(env, `/v1/files/${encodeURIComponent(object_id)}/complete`);\n if (!done.ok) return new Response(`complete ${done.status}`, { status: 502 });\n\n // 3. settle the render, on the same keyless callback the container uses\n const settled = await fetch(t.callback_url, {\n method: 'POST', headers: { 'content-type': 'application/json' },\n body: JSON.stringify({ status: 'completed', output_file: object_id, progress: 100, stage: 'done' }),\n });\n return settled.ok ? Response.json({ object_id }) : new Response(`callback ${settled.status}`, { status: 502 });\n }\n return new Response('not found', { status: 404 });\n },\n};\n\ndeclare function launchContainer(job: Record<string, unknown>): Promise<void>; // your container platform's API\n```\n\nThe container's side is three kinds of HTTP call and no vxil key: `processing` pings to\n`callback_url`, the file to `output_url` (with `Authorization: Bearer <output_ticket>` and the\nfile's `Content-Type`), and `{\"status\": \"failed\", \u2026}` to `callback_url` if it gives up. Four notes\non the shape:\n\n- **A retried output post** (the container saw a network error after the coordinator had uploaded)\n mints a second object, and the second `completed` is answered with the settled state and changes\n nothing. Delete the extra object, or keep a `render_id \u2192 object_id` note in the coordinator's own\n store and skip the upload when one exists.\n- **Very large outputs.** The coordinator above holds the whole file in memory, and passing gigabytes\n through it costs its bandwidth too. For big files, have the container report the size first\n (`POST /output-url` with its ticket on the `Authorization` header and `{ size_bytes }`) and let the coordinator answer with the\n presigned upload URL it minted; the container PUTs straight to it, and the coordinator completes\n the object and posts `completed` when the container says it is done. The container still holds no\n key: a presigned URL is good for one object for a few minutes.\n- **Upgrading an earlier copy of this blueprint.** Renders started before `request-render` put\n `user_id` in the start body have none, and a server-mode upload must name its user. The\n coordinator refuses such a start with a `400`, which ends that run and refunds it at once;\n redeploy `request-render` (`vxil push`) before you point `render_url` at the coordinator.\n- **Serving it.** The row now holds a files object id. Read it back with a signed download URL, or\n publish it from a settle function with a server key (`vx.files.publish(object_id)`) for a stable\n public URL served from the edge cache.\n\nOptions A and B below show the same contract with the runtime posting `completed` itself (an\n`output_url` in your own bucket). Either can adopt the coordinator: point `render_url` at it, and have\nstep 1 trigger the Trigger.dev task or spawn the Modal function.\n\n## Option A \u2014 Trigger.dev (v4)\n\nTrigger.dev runs the render as a task on a machine you pick, with ffmpeg or Chromium baked into the\nimage, retries, and its own run dashboard. Two pieces: a **relay endpoint** that turns vxil's start\nrequest into a Trigger.dev trigger, and the **task**.\n\n**Why a relay, and not `render_url` pointed straight at Trigger.dev's trigger API?** vxil sends\n`callback_url` beside `payload` at the top level of the body, and Trigger.dev's trigger API passes\nonly `payload` to the task \u2014 the task would never see where to report. The relay is ~30 lines and is\nalso where your `render_token` is checked. Host it anywhere that serves https: a serverless function on\nyour web host, a small edge worker, a route in your existing API.\n\n```ts\n// relay.ts \u2014 your render_url. Standard fetch handler (edge worker / serverless function / Node 18+ adapter).\n// Env: RENDER_TOKEN (the same value as the vxil secret render_token), TRIGGER_SECRET_KEY (tr_prod_\u2026 / tr_dev_\u2026).\ntype Start = {\n render_id: string; composition: string; props?: Record<string, unknown>;\n payload: { generation_id: string; correlation_id?: string; deadline_at: string }; callback_url: string;\n};\n\nexport default {\n async fetch(req: Request, env: { RENDER_TOKEN: string; TRIGGER_SECRET_KEY: string }): Promise<Response> {\n if (req.method !== 'POST') return new Response('method not allowed', { status: 405 });\n if (req.headers.get('authorization') !== `Bearer ${env.RENDER_TOKEN}`) {\n return new Response('unauthorized', { status: 401 }); // a 4xx ends the vxil run (and refunds)\n }\n const s = (await req.json()) as Start;\n const res = await fetch('https://api.trigger.dev/api/v1/tasks/render-video/trigger', {\n method: 'POST',\n headers: { authorization: `Bearer ${env.TRIGGER_SECRET_KEY}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n payload: {\n render_id: s.render_id, composition: s.composition, props: s.props ?? {},\n callback_url: s.callback_url, deadline_at: s.payload.deadline_at,\n },\n options: {\n // a start vxil re-sends (a lost answer) triggers the SAME Trigger.dev run\n idempotencyKey: `render:${s.render_id}`,\n // a render still queued after 3 minutes is dropped (it never runs, and no\n // onFailure fires \u2014 vxil refunds it at its deadline). Part of the budget below.\n ttl: '3m',\n tags: [`render_${s.render_id}`],\n },\n }),\n });\n if (res.ok) return Response.json({ accepted: true }, { status: 202 });\n // Trigger.dev busy or down: answer 503 and vxil tries the start again; anything else ends the run\n return new Response(`trigger.dev ${res.status}`, { status: res.status === 429 || res.status >= 500 ? 503 : 400 });\n },\n};\n```\n\n```ts\n// trigger.config.ts \u2014 ffmpeg and Chromium in the task image\nimport { defineConfig } from '@trigger.dev/sdk';\nimport { ffmpeg } from '@trigger.dev/build/extensions/core';\nimport { puppeteer } from '@trigger.dev/build/extensions/puppeteer';\n\nexport default defineConfig({\n project: '<your project ref>',\n dirs: ['./trigger'],\n maxDuration: 600, // CPU seconds PER ATTEMPT \u2014 the task sets its own; see the budget below\n build: { extensions: [ffmpeg(), puppeteer()] }, // puppeteer also needs PUPPETEER_EXECUTABLE_PATH set in the Trigger.dev env\n});\n```\n\n```ts\n// trigger/render-video.ts \u2014 the task: render, upload, report to vxil\nimport { task, metadata, logger } from '@trigger.dev/sdk';\n\ntype Payload = {\n render_id: string; composition: string; props: Record<string, unknown>;\n callback_url: string; deadline_at: string; // when vxil stops waiting (ISO)\n};\n\n/** The longest one attempt takes, wall clock, with margin. An attempt that cannot\n * finish before deadline_at does not start: it tells vxil, so the credits come back now. */\nconst ATTEMPT_WALL_MS = 12 * 60_000;\n\n/** POST one status to vxil. A 5xx/network error throws, so Trigger.dev retries the attempt;\n * a repeated completed/failed post is a no-op on vxil's side. */\nasync function report(callbackUrl: string, body: Record<string, unknown>): Promise<void> {\n const res = await fetch(callbackUrl, {\n method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body),\n });\n if (res.status >= 500) throw new Error(`callback to vxil answered ${res.status}`);\n}\n\nexport const renderVideo = task({\n id: 'render-video',\n machine: 'large-1x', // 4 vCPU / 8 GB \u2014 size to your renders\n maxDuration: 600, // CPU seconds per attempt (10 min) \u2014 not wall time\n retry: { maxAttempts: 2, minTimeoutInMs: 5_000, maxTimeoutInMs: 30_000 },\n run: async (p: Payload) => {\n if (Date.now() + ATTEMPT_WALL_MS > Date.parse(p.deadline_at)) {\n // too late to finish inside vxil's deadline: refund now, and do not retry\n await report(p.callback_url, { status: 'failed', error: 'deadline', hint: 'no time left for another attempt' });\n return { skipped: 'deadline' };\n }\n metadata.set('stage', 'rendering'); // Trigger.dev's own run view\n await report(p.callback_url, { status: 'processing', stage: 'rendering', progress: 0 });\n\n // \u2026your render: drive Chromium for frames, run ffmpeg, write the file\n // to YOUR bucket keyed by render_id (so a retried attempt overwrites, not duplicates)\u2026\n const outputUrl = `https://cdn.example.com/renders/${p.render_id}.mp4`;\n const durationS = 31.2;\n logger.info('rendered', { render_id: p.render_id, outputUrl });\n if (Date.now() > Date.parse(p.deadline_at)) {\n // vxil has already failed and refunded this render: the post below is answered\n // with that settled state. Your sizing is off \u2014 widen the budget below.\n logger.warn('finished after the vxil deadline', { render_id: p.render_id });\n }\n\n await report(p.callback_url, {\n status: 'completed', output_url: outputUrl, duration_s: durationS, progress: 100, stage: 'done',\n });\n return { output_url: outputUrl };\n },\n // after the last attempt THROWS: tell vxil, so the credits come back now, not at the\n // deadline. Not called when an attempt exceeds maxDuration or the run expires on its\n // ttl \u2014 those refund only at vxil's deadline.\n onFailure: async ({ payload, error }) => {\n await report(payload.callback_url, {\n status: 'failed', error: 'render_failed', hint: String(error instanceof Error ? error.message : error).slice(0, 200),\n });\n },\n});\n```\n\n**Budget the wall clock.** Trigger.dev's limits and vxil's deadline are separate clocks, and only\nvxil's refunds. Size them so a render always ends \u2014 `completed` or `failed` \u2014 before vxil's deadline:\n\n```\nttl + maxAttempts \xD7 (longest attempt, wall clock) + retry backoff < timeout.after_ms\n3 min + 2 \xD7 12 min + \u2264 1 min = 28 min < 30 min\n```\n\n`maxDuration` counts **CPU time per attempt**, not wall time across the run, so it does not bound\nthe sum: the `deadline_at` check at the start of each attempt does. If your renders need more, raise\n`RENDER_DEADLINE_MS` in `request-render` (up to `generation.maxTimeoutMs`, one hour) and resize the\nrest to fit.\n\n**Be honest with yourself about three things before you ship on Trigger.dev Cloud:**\n\n- **Data residency.** Trigger.dev Cloud keeps its operational and log data \u2014 including each run's\n payload \u2014 in the US (us-east-1), even when the machines run elsewhere. Your render props and the\n `callback_url` pass through it. If that rules it out, self-host Trigger.dev for your own app, or use\n option B.\n- **The callback URL is a credential for one render.** It appears in Trigger.dev's run payload and\n dashboard. It can settle only that render, and stops mattering once the render is settled.\n- **Two clocks, and `onFailure` is not a guarantee.** Trigger.dev calls `onFailure` only after the last\n attempt throws. A run that exceeds `maxDuration`, or expires on its `ttl` before it starts, ends\n without it \u2014 vxil refunds those at its deadline, not sooner. Keep the budget above, and keep the\n `deadline_at` check, so a late attempt refunds early instead of finishing after the refund.\n\n## Option B \u2014 your own container runtime (Modal, Fly, a container on your own cloud account)\n\nSame contract, no relay: the endpoint you deploy **is** `render_url`. It must answer within 20 seconds,\nso it only checks the token, hands the job to a background worker and answers `2xx`; the worker renders\nand posts back. On Modal:\n\n```python\n# render_app.py \u2014 `modal deploy render_app.py`; render_url = the endpoint's https URL\nimport json, os, urllib.request\nfrom datetime import datetime, timedelta, timezone\nimport modal\nfrom fastapi import HTTPException, Request # also `pip install fastapi` where you run `modal deploy`\n\nimage = (modal.Image.debian_slim()\n .apt_install(\"ffmpeg\", \"chromium\")\n .pip_install(\"fastapi[standard]\"))\napp = modal.App(\"render-farm\", image=image)\nsecrets = [modal.Secret.from_name(\"render-farm\")] # RENDER_TOKEN\n\ndef report(callback_url: str, body: dict) -> None:\n req = urllib.request.Request(callback_url, data=json.dumps(body).encode(),\n headers={\"content-type\": \"application/json\"}, method=\"POST\")\n urllib.request.urlopen(req, timeout=30)\n\nATTEMPT_WALL_S = 25 * 60 # = the timeout below; a call cut off there may never reach its except\n\n@app.function(cpu=4, memory=8192, timeout=ATTEMPT_WALL_S, secrets=secrets)\ndef render(job: dict) -> None:\n cb = job[\"callback_url\"]\n deadline = datetime.fromisoformat(job[\"payload\"][\"deadline_at\"].replace(\"Z\", \"+00:00\"))\n if datetime.now(timezone.utc) + timedelta(seconds=ATTEMPT_WALL_S) > deadline:\n # queued too long to finish before vxil stops waiting: refund now\n report(cb, {\"status\": \"failed\", \"error\": \"deadline\", \"hint\": \"started too late to finish\"})\n return\n try:\n report(cb, {\"status\": \"processing\", \"stage\": \"rendering\", \"progress\": 0})\n # \u2026render with ffmpeg / chromium, upload to YOUR bucket keyed by job[\"render_id\"]\u2026\n output_url = f\"https://cdn.example.com/renders/{job['render_id']}.mp4\"\n report(cb, {\"status\": \"completed\", \"output_url\": output_url, \"duration_s\": 31.2,\n \"progress\": 100, \"stage\": \"done\"})\n except Exception as e: # tell vxil now, so the credits come back before the deadline\n report(cb, {\"status\": \"failed\", \"error\": \"render_failed\", \"hint\": str(e)[:200]})\n raise\n\n@app.function(secrets=secrets)\n@modal.fastapi_endpoint(method=\"POST\")\nasync def start(request: Request):\n if request.headers.get(\"authorization\") != f\"Bearer {os.environ['RENDER_TOKEN']}\":\n raise HTTPException(status_code=401, detail=\"unauthorized\") # a 4xx ends the vxil run\n job = await request.json()\n await render.spawn.aio(job) # queued; returns at once, well inside the 20-second window\n return {\"accepted\": True}\n```\n\n`spawn` queues the call and returns immediately, so the endpoint answers in well under a second. A\nstart can arrive twice (a lost answer is retried), so dedupe on `render_id`: keep the ids you have\nspawned in a `modal.Dict`, or make the render overwrite the same output key.\n\nAnything else that can (1) answer an https POST in under 20 s, (2) run the work in the background and\n(3) POST JSON to a URL fits the same contract: a Fly Machine started per job, a container service with\na queue in front, your own GPU box. vxil does not care what runs the render \u2014 only that the start is\nacknowledged quickly and the callback eventually comes.\n\n## Run it\n\nGive a user some credits from your server (or sell `render_pack_100` through your payments provider):\n\n```bash\ncurl -s -X POST \"https://api.vxil.com/v1/payments/credits/grant\" \\\n -H \"authorization: Bearer $KEY\" -H 'content-type: application/json' \\\n -H 'idempotency-key: welcome-u1' \\\n -d '{\"user_id\":\"<the user id>\",\"credit_type\":\"render_credits\",\"amount\":25,\"source\":\"welcome\"}'\n```\n\nStart a render **with the user's session** (end-user mode \u2014 the held credits are forced onto that\nuser):\n\n```ts\nimport { Vxil } from '@vxil/sdk';\n\n// after `vxil gen`, vx.fn['request-render'] is typed from the function's declared signature\nconst vx = new Vxil({ apiKey: process.env.VXIL_PUBLISHABLE_KEY!, endUserToken: process.env.USER_SESSION! });\nconst started = await vx.fn['request-render']({\n composition: 'promo-30s', request_key: 'promo-1', props: { headline: 'Spring sale' },\n});\n// \u2192 { item_id, run_id, credits: 5 }\n// (or { duplicate: true, request_key, item_id, run_id, status } on a retry)\n```\n\nThen read the row \u2014 or subscribe to its changes \u2014 until `status` is `completed`:\n\n```ts\nif ('item_id' in started) {\n const row = await vx.from('renders').get(started.item_id);\n // row.status \u2192 'completed', row.output_url \u2192 'https://cdn.example.com/renders/\u2026.mp4'\n}\n```\n\n**Try it before you have a runtime.** Point `render_url` at any https endpoint that answers `2xx`\n(a request-bin works) and play the runtime yourself: copy `callback_url` from the request it received,\nthen\n\n```bash\ncurl -s -X POST \"$CALLBACK_URL\" -H 'content-type: application/json' \\\n -d '{\"status\":\"completed\",\"output_url\":\"https://cdn.example.com/x.mp4\",\"duration_s\":12.5,\"progress\":100,\"stage\":\"done\"}'\n# \u2192 { \"data\": { \"run_id\": \"run_\u2026\", \"generation_status\": \"completed\" } } \u2014 and the row says so\n```\n\n## How it fails, and what the user sees\n\n| what happened | the run | the row | the credits |\n|---|---|---|---|\n| runtime posted `completed` | `completed` | `status: completed` + every key it sent | committed |\n| runtime posted `failed` | `failed` | `status: failed`, `error` = its code + hint (written by `notify-ready`) | refunded |\n| runtime never called back | `failed` (`GenerationExpired`) at the deadline | `failed`, `error: GenerationExpired: the render did not finish before its deadline` | refunded |\n| runtime finished after the deadline | already `failed`; the late `completed` is answered with that state | `failed` | refunded (your compute was spent \u2014 budget the clocks) |\n| your endpoint answered `5xx` / timed out | start retried with backoff; terminal after the attempts | `processing` \u2192 `failed`, `error: RetryableHttp: the render endpoint kept failing to accept the render (retries exhausted)` (or `NetworkError: \u2026` when it could not be reached) | held until then, then refunded |\n| your endpoint answered another `4xx` (a bad token) | `failed` at once | `failed` | refunded |\n| the user had too few credits (at `request-render`) | ended at once (`ReserveInsufficient`), never started | `failed`, `error: insufficient_credits`; that `request_key` is spent; the caller got the `402`, no message | nothing held |\n| too many renders in flight, or a jobs-side fault | not created yet (the caller gets `429` with `queued: true` and `retry_after`, or `502` with `queued: true`) | `pending`, no run \u2014 **queued**: `redrive-pending` starts it when a slot frees up (or the app calls again with the **same** `request_key`) | held when it starts |\n| the user had too few credits when the re-driver started it | ended at once (`ReserveInsufficient`) | `failed`, `error: insufficient_credits`; the owner is told once (they last heard \"queued\") | nothing held |\n| still no free slot an hour later | never created | `failed`, `error: not_started: the render waited too long for a free slot` (written by `redrive-pending`); the owner is told once | nothing held |\n| the re-driver's start was refused (`400` / `401` / `403` / `422`: a non-https or private `render_url`, a wrong or revoked `vxil_jobs_key`) | not created | still `pending` \u2014 the tick stops and reports it (`stopped: { status, code }` in the function's logs); fix the setup and the next tick carries on; the one-hour bound still applies | nothing held |\n\n## The backlog: `maxConcurrent`, `429` and the re-driver\n\n`generation.maxConcurrent` is how many generation runs this project may have **open** at once: 20 by\ndefault, settable up to 200 in the jobs config. This blueprint sets 20; raise it to what your plan\nand your runtime can carry:\n\n```ts\njobs: { enabled: true, generation: { maxConcurrent: 50 /* 1\u2013200, default 20 */ } },\n```\n\nAt the cap a new start is refused with `429` and `Retry-After: 5`. **No run is created and nothing\nis held yet.** `request-render` passes the `429` on to the app with `queued: true` (and the\n`item_id` and `request_key`), and leaves the row `pending` with no `run_id`. That render is\n**queued, not refused**: the re-driver will start it, and hold its credits then, up to an hour\nlater. So the app must treat a `429` with `queued: true` as \"queued\" \u2014 show it, watch the row \u2014 and\nmust **never retry it with a new `request_key`**: that is a second render, and both are charged. A\nretry with the **same** `request_key` is always safe. Rows like that are the backlog, and two\nthings drain it:\n\n1. **`redrive-pending`, every minute.** It reads the oldest `pending` rows with no run that are at\n least 30 s old (`{ status: 'pending', created_at: { $lt: \u2026 }, run_id: null }`, sorted by\n `created_at`, 20 per tick; `status` and `created_at` are index slots, so the read stays cheap at\n any size) and starts each one with the same descriptor and the **same idempotency key** as\n `request-render`, so a row the app is re-driving at the same moment still gets one run. It\n **stops at the first `429`**, since the rest of the batch would get the same answer, and the next\n tick carries on. Each try is recorded on the row (`redrive_attempts`, `redriven_at`). A `402`\n fails the row with `insufficient_credits`. A `400`, `401`, `403` or `422` also **stops** the tick\n and fails nothing: every re-driven row sends the same descriptor, so a refusal means the setup is\n wrong (a non-https or private `render_url`, a revoked key), not the row; the tick reports it as\n `stopped: { status, code }`. The only thing that fails a waiting row is age: a row still waiting\n after **one hour** is failed with `not_started: the render waited too long for a free slot`.\n Nothing was ever held, so nothing is refunded. In both the `402` and the one-hour case the owner\n is told once (`Idempotency-Key: render-not-started:<item_id>`), because the last thing they heard\n was \"queued\". (The function's scopes include `notifications:send` for that.)\n2. **The app**, calling `request-render` again with the same `request_key`, if it wants the render\n started sooner than the next tick.\n\n`overlap: 'skip'` keeps a slow tick from being doubled by the next one. On the Free plan, run it\nevery 15 minutes (see the plan note at the top).\n\n**Why it has its own key.** A credit hold placed by a function must name the signed-in user the\nfunction acts for, and a cron tick has none. So `redrive-pending` makes the start with\n`vxil_jobs_key`, an API key holding only `jobs:write`, the way your trusted server would. That is\nmore than \"hold credits and start runs\": `jobs:write` also lets the key cancel or replay any run of\nthis project, enqueue any job, and create or change schedules and flow rules, and it can hold\ncredits against any of your users and start runs against any public https endpoint. Treat it as a\nserver key: keep it only in this function's secrets, never in an app, and rotate it like any\nserver key. The re-driver never reads the\nprice from the row (a signed-in user can edit their own row): the hold is the function's fixed\n`RENDER_CREDITS`, and a row whose `render_key` is not `owner:request_key` is failed, not started.\n(A hand-written row like that which also breaks the `render_key_shape` hook cannot be written at\nall, so the tick counts it as `unwritable` and it keeps one of the 20 slots: fix or delete it by\nhand.)\n\n## A contract test for your runtime\n\nEvery key your runtime posts in the `completed` body is written onto the row, so every key must be a\ndeclared field of `renders`. Keep that true in your own CI with a test beside your runtime's code. It\nfails the day someone adds a key to the completion body and forgets the field:\n\n```ts\n// render-contract.test.ts \u2014 vitest, in YOUR repo (the one holding vxil.config.ts)\nimport { describe, expect, it } from 'vitest';\nimport config from './vxil.config';\n// the bodies your runtime (or coordinator) really posts: import the builders from that code,\n// so the test follows it instead of a hand-copied list\nimport { completedBody, progressBody } from './runtime/callback-bodies';\n\ndescribe('render callbacks', () => {\n const declared = new Set(Object.keys(config.cms!.collections!.renders!.fields!));\n\n it('every key of the completion body is a declared renders field', () => {\n const body = completedBody({ objectId: 'obj_test', durationS: 1 });\n expect(Object.keys(body).filter((k) => !declared.has(k))).toEqual([]);\n });\n\n it('a progress ping carries only the keys the mirror keeps', () => {\n const ping = progressBody({ progress: 40, stage: 'encoding' });\n expect(Object.keys(ping).filter((k) => !['status', 'progress', 'stage', 'message'].includes(k))).toEqual([]);\n });\n});\n```\n\nThe platform also has a fallback for the day that test is missing. When the row refuses a completion\nmirror because of an undeclared key, vxil writes the status alone, so the row still says\n`completed`, and the run reports what it dropped (`GET /v1/jobs/runs/{run_id}` \u2192 `mirror_error`,\nwith `fields_dropped`). The app no longer hangs on `processing`, but the dropped keys are not on the\nrow. The test is what keeps them there.\n\n## The bounds to design against\n\n- **Deadline**: the run's timeout is clamped to `generation.maxTimeoutMs` \u2014 one hour at most. A render\n that can take longer should be split (the next step starts from `notify-ready`), or tracked on your own\n row without a platform-held reserve.\n- **In flight**: `generation.maxConcurrent` renders at once (20 here and by default; up to 200). Over\n it, `request-render` answers `429` and the row waits for `redrive-pending`, or for a retry with the\n same `request_key` ([the backlog](#the-backlog-maxconcurrent-429-and-the-re-driver)).\n- **Holds**: one render's reserve is clamped to `generation.maxReserveCredits` (50 here), and the sum of\n all open holds is capped by `generation.maxOutstandingReserveCredits` (5,000 here).\n- **Callback**: at most 256 KiB per post, JSON, every key a declared field of `renders`.\n\n## Your token, and where it lives\n\n`render_url` and `render_token` are function secrets; `request-render` reads them at invoke time and\nputs them on the generation run, which vxil stores with the run (your project only) for as long as the\njobs retention keeps it. The token is **never returned by a read**: `GET /v1/jobs/runs/{run_id}` (and\nthe dashboard and MCP reads built on it) shows the provider header names with every value as\n`[redacted]`. Rotate it in both places (`vxil secrets set functions/render_token` and your endpoint's\nenv); runs already started keep the token they were started with.\n\n## Evidence\n\n- **Read in vxil's code**: the start request body (`provider.body` + `payload` + `callback_url`), the\n 20-second start bound, the retry ladder, the status mirror writing every completion key onto the row,\n the hold committed on `completed` and released on `failed` / the deadline, and `error` / `hint`\n becoming `error_class` / `error_hint` on `job.generation.failed`.\n- **Read in Trigger.dev's and Modal's documentation, not executed from vxil**: the trigger endpoint\n `POST https://api.trigger.dev/api/v1/tasks/{taskId}/trigger` with `{ payload, options }` and the\n `idempotencyKey` / `ttl` / `tags` / `machine` options; `task({ id, machine, maxDuration, retry, run,\n onFailure })` and `retry` options, `maxDuration` being CPU time per attempt with no `onFailure` when\n it is exceeded, `metadata.set`, and the `ffmpeg()` / `puppeteer()` build extensions; Trigger.dev\n Cloud's US-hosted operational data; Modal's `@modal.fastapi_endpoint` and `.spawn()`. Check each\n vendor's current docs before you ship.\n",
18565
+ "readme": "# Render Farm \u2014 long renders on a runtime you rent, with vxil holding the credits, the deadline and the callback\n\n```bash\nmkdir my-renders && cd my-renders\nvxil init --template render-farm # init scaffolds into the CURRENT directory\nprintf '%s' \"$RENDER_URL\" | vxil secrets set functions/render_url # your render endpoint (https)\nprintf '%s' \"$RENDER_TOKEN\" | vxil secrets set functions/render_token # a long random token it checks\n# the backlog re-driver's key: an API key of this backend holding ONLY jobs:write\nvxil login # keys mint needs a dashboard session, not a project key\nvxil keys mint --name render-redrive --scopes jobs:write --json | jq -r .api_key | vxil secrets set functions/vxil_jobs_key\nvxil push\n```\n\n> **Plan note.** The functions deploy on the Free plan when the project's workload is `staging` or\n> `development` (`vxil projects workload <slug> development`, or create it with\n> `vxil projects create <slug> --workload development`). On a Free `production` project, `vxil push` stops before it writes anything, naming the plan and the ways out: change the workload or upgrade to Developer, or run `vxil push --skip-functions` to apply the collections and config without the functions.\n> On the Free plan a function cron may also fire at most every 15 minutes, so change `redrive-pending`'s\n> schedule to `'*/15 * * * *'` there. The blueprint is written for Developer and up, where it runs every\n> minute. Everything else works the same; a backlog just drains more slowly.\n\nA video render, a transcode, a headless-browser capture: minutes of CPU, ffmpeg or Chromium. That does\nnot fit in a vxil function (a delivered trigger gets about a minute), and vxil will not grow a container\ntier or a workflow engine to run it. So the work runs on **a runtime you rent** \u2014 Trigger.dev, Modal,\na container on your own cloud account \u2014 and vxil keeps the four things that must survive while it\nruns:\n\n| vxil holds | so that |\n|---|---|\n| **the credits** reserved for the render | a failed, abandoned or cancelled render gives them back, and a user can never start more than they can pay for |\n| **the deadline** | a render your runtime never reports on fails and refunds after 30 minutes (at most one hour) |\n| **the signed completion callback** | your runtime needs no vxil key: the URL it is handed is the credential for that one render |\n| **the status row** | the app reads (or subscribes to) one `renders` row: `pending \u2192 processing \u2192 completed | failed`, plus everything your runtime sent back |\n\n**What this blueprint teaches that the others do not:** the hand-off to **your own** long-running\nruntime through a **webhook-mode generation run** \u2014 the contract your endpoint and your worker must\nkeep, and two complete runtime options below. (`fal-media` shows the same lane against a vendor queue\nAPI; `job-runner` shows a provider call vxil polls.)\n\n## What you get\n\n- **`renders`** \u2014 one row per render, owned by the user who asked for it (`strictEndUserScope`: a\n signed-in user reads only their own). `render_key` is the owner + `:` + the client's `request_key`\n (the composition is enforced by a `beforeWrite` hook) and is **unique**, so a double tap or a retried\n request finds the first row instead of starting a second render. Your runtime's answer lands on the\n row: `output_url`, `duration_s`, `progress`, `stage`, `message`.\n- **`request-render`** (http function, end-user mode) \u2014 creates-or-finds the row, then starts ONE\n generation run: your endpoint (`render_url`), your token on its `Authorization` header, a status\n mirror onto the row, `reserve_credits` for the render (5 `render_credits`), a 30-minute deadline, and\n the `render_key` as the run's `idempotency_key` \u2014 so a re-driven start gets the same run back. The\n credits it holds are the function's fixed price, never a value read from the row.\n- **`redrive-pending`** (cron function, every minute, `overlap: 'skip'`) \u2014 drains the backlog. When\n the project already has `generation.maxConcurrent` renders in flight, `request-render` answers `429`\n (with `queued: true`) and the row waits `pending` with no run. This function starts those rows,\n oldest first, with the **same** idempotency key, and tells the owner when one never starts\n ([the backlog](#the-backlog-maxconcurrent-429-and-the-re-driver)).\n- **`notify-ready`** (webhook function on `job.generation.`) \u2014 writes the failure cause onto the row (a\n platform class such as `GenerationExpired` gets a short human hint after it) and sends the owner one\n message per run (`Idempotency-Key: render-ready:<run_id>`). A render refused for too few credits\n keeps `insufficient_credits`; when `request-render` refused it, the caller already got the `402`\n and nothing is sent, and when the re-driver started it (the user last heard \"queued\"), the owner\n is told once (`Idempotency-Key: render-not-started:<item_id>`, the re-driver's own key). It\n **branches on `error_class`**, because not every `job.generation.failed` is a render that failed:\n `PaymentsUnavailable` (the credits could not be held at the start \u2014 the row is still queued and a\n new run follows) is skipped (and when a re-sent `request-render` raced that start and linked its\n run, the row is unlinked \u2014 `run_id: null`, only while it is still `pending` on that run \u2014 so\n `redrive-pending` starts it), and `Cancelled` (your own cancel) writes `error: cancelled` and sends\n nothing. Both are also lower-level events (`warn` / `info`), so they stay out of an immediate\n failure digest.\n- **credits** \u2014 a payments integration on the `mock` provider (no provider account needed to try it).\n \"Credits\" are usage units you meter, not money.\n\nThe row is a **view** for the app; the run and the ledger are the truth. If your app's client key\ncarries `cms:write`, a signed-in user can edit their own `renders` row (say, set `status` to\n`completed`) \u2014 that changes nothing they are charged or given. Anything that grants something on\ncompletion should read the run (`GET /v1/jobs/runs/{run_id}`) or react to `job.generation.completed`,\nas `notify-ready` does \u2014 or give the client key only `cms:read`.\n\n## The contract your runtime keeps\n\nWhatever runs the render, these are the only things it has to do:\n\n| Step | What arrives / what to send |\n|---|---|\n| **Start** | vxil `POST`s your `render_url` with `Authorization: Bearer <render_token>`, the header `x-vxil-run-id` and JSON `{ render_id, user_id, composition, props, payload: { generation_id, correlation_id, deadline_at }, callback_url }` (`user_id` is the render's owner). Check the token, **store or queue the work and answer `2xx` within 20 seconds** \u2014 never render inline, and never wait on a container's cold start. When you cannot take the work right now, say so at once with a `503` instead of holding the request open; the start is then sent again later, as after a timeout. `408` / `429` / `5xx` / a timeout is retried with backoff (30 s, then 1, 2 and 4 minutes, doubling up to 10 minutes; a `Retry-After` on your answer sets the wait, 1 s to 10 min, never past the deadline; `max_attempts` counts start calls) until the run's attempts are used \u2014 with the default five, the last try comes 7 to 8 minutes after the first, so leave room for that in the deadline. Any other `4xx` ends the run and refunds the credits. A start can arrive more than once (a lost answer is retried), so **dedupe by the run id** (`x-vxil-run-id`; in this blueprint `render_id` = `payload.generation_id` names the same single run), never by a hash of the content ([why](#one-render-one-run-dedupe-a-doubled-start)). |\n| **Progress** (optional, best-effort) | `POST callback_url` with `{\"status\": \"processing\", \"progress\": 40, \"stage\": \"encoding\"}`. A processing ping moves the row to `processing` and writes its `progress` (0\u2013100) / `stage` (\u2264 64 chars) / `message` (\u2264 200 chars) onto the row \u2014 `request-render` asks for that with `status_mirror.progress_fields` \u2014 so the app shows live progress by watching the row. Other keys on a ping are not written to the row (the run keeps the latest report, `GET /v1/jobs/runs/{run_id}` \u2192 `progress`). **At most one ping every 5 seconds**, sending the latest state; a ping that fails or answers `429` is simply dropped ([callback limits](#callback-limits-progress-is-best-effort-the-final-post-must-arrive)). |\n| **Done** (must arrive) | `POST callback_url` with `{\"status\": \"completed\", \"output_url\": \"https://\u2026\", \"duration_s\": 31.2, \"progress\": 100, \"stage\": \"done\"}` \u2014 or, when the output is uploaded into this project's files, `{\"status\": \"completed\", \"output_file\": \"obj_\u2026\", \"duration_s\": 31.2, \"progress\": 100, \"stage\": \"done\"}` ([below](#keys-stay-out-of-the-container-the-recommended-shape)). At most 256 KiB, and **every key a declared field of `renders`** (add a field before you send a new key; [test it](#a-contract-test-for-your-runtime)). The credits are committed and every key is written onto the row. **Retry this post on `429`, `5xx` and network errors, honouring `Retry-After`, until `deadline_at`** \u2014 never drop it (`postCallback` below). |\n| **Failed** (must arrive) | `POST callback_url` with `{\"status\": \"failed\", \"error\": \"render_failed\", \"hint\": \"ffmpeg exited 1: \u2026\"}`. The credits are refunded; `error` (a short code) and `hint` reach `notify-ready` as `error_class` / `error_hint` and land on the row's `error`. Retried exactly like `completed`. |\n| **Never answers** | the run fails at the deadline (`GenerationExpired`), the credits are refunded, and a callback after that changes nothing. |\n| **The deadline** | `payload.deadline_at` (an ISO time) is when vxil stops waiting \u2014 to the second: any callback that arrives at or after it ends the run as `GenerationExpired` (credits refunded), and a `completed` posted then is answered `{ generation_status: \"failed\", expired: true }` \u2014 the output exists, but the user was refunded and the row says failed. Check it before each attempt starts: a render that cannot finish by then should post `failed` and stop. |\n\n`callback_url` needs no other credential \u2014 and nothing else should see it. A repeated `completed` or\n`failed` post is answered with the settled state and changes nothing, so your worker can safely retry\nits own callback on a network error. The output bytes stay where your runtime wrote them (your bucket,\nyour CDN) and vxil stores the keys, not the file, unless a coordinator uploads the output into this\nproject's files ([next section](#keys-stay-out-of-the-container-the-recommended-shape)).\n\nEach start request also carries an `X-Vxil-Jobs-Signature` header (verifiable with your project's\njobs signing secret, `GET /v1/jobs/signing-secret`) and `x-vxil-run-id`. This blueprint uses the\nbearer token because it is one string comparison in any language.\n\n### Callback limits: progress is best-effort, the final post must arrive\n\nEach render's `callback_url` has its **own** budget at vxil's edge: about 120 posts a minute for that\none run. Every post over it, pings included, is counted against a small separate allowance (about 20\na minute), and on that allowance only a `completed` or `failed` post is accepted. A runtime that pings\nevery 5 seconds never gets near either number. One that sends more than about 140 posts in a minute\nuses the allowance up as well, and its final post waits for the next minute. All of a project's\ncallbacks together also share a ceiling of about 3,000 a minute, plus about 300 a minute for final posts.\nEvery number here is best-effort (counted per serving machine). The signature in the URL identifies\nthe run, so fifty renders posting from one shared egress IP do not share a budget. Over a budget the\nanswer is `429 rate_limited` with `Retry-After: 30`.\n\nTwo rules follow, and the helpers below keep both:\n\n- **Progress is best-effort.** Post at most one `processing` ping every 5 seconds, carrying the latest\n state; a ping that fails or answers `429` is dropped, and the next one carries the newer state.\n- **The final post must arrive.** A `completed` or `failed` post that answers `429`, `5xx` or never\n gets an answer is sent again, after `Retry-After` when there is one, until `deadline_at`. A repeat\n of a final post is harmless: it is answered with the settled state and changes nothing. Any other\n `4xx` means the URL or the body is wrong, so retrying will not help.\n\n```ts\n// callbacks.ts \u2014 for the coordinator, a task, or any worker that posts to callback_url\n/** completed / failed: MUST arrive. Retries 429, 5xx and network errors (honouring Retry-After)\n * until `untilMs`; throws only when it cannot deliver in time, or on another 4xx. */\nexport async function postCallback(callbackUrl: string, body: Record<string, unknown>, untilMs: number): Promise<void> {\n for (let attempt = 0; ; attempt++) {\n let res: Response | undefined;\n try {\n res = await fetch(callbackUrl, {\n method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body),\n });\n } catch { /* a network error: send it again */ }\n if (res && res.ok) return;\n if (res && res.status !== 429 && res.status < 500) throw new Error(`callback refused: ${res.status}`);\n const retryAfterS = Number(res?.headers.get('retry-after'));\n const waitMs = retryAfterS > 0 ? retryAfterS * 1000 : Math.min(1000 * 2 ** attempt, 30_000);\n if (Date.now() + waitMs > untilMs) throw new Error(`callback not delivered in time (last answer: ${res?.status ?? 'none'})`);\n await new Promise((r) => setTimeout(r, waitMs));\n }\n}\n\n/** processing: best-effort, at most one post every `gapMs`; a refused or failed ping is dropped. */\nexport function progressReporter(callbackUrl: string, gapMs = 5_000) {\n let last = 0;\n return async (p: { progress?: number; stage?: string; message?: string }): Promise<void> => {\n if (Date.now() - last < gapMs) return; // coalesced: the next ping carries newer state\n last = Date.now();\n await fetch(callbackUrl, {\n method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify({ status: 'processing', ...p }),\n }).catch(() => undefined);\n };\n}\n```\n\n### One render, one run: dedupe a doubled start\n\nvxil sends the start again when it did not get your answer: a timeout, a dropped connection, a `5xx`.\nSo the same start can reach you twice, even while the first copy is still being handled. Claim the\nwork under the **run id** (`x-vxil-run-id`) before you launch anything. Answer `202` to a repeat\nonly once the work is launched. While the first copy is still launching, answer `503`, so the start\nstays open and vxil sends it again; a `202` would end vxil's retries even if that first launch then failed. In this blueprint `render_id` (= `payload.generation_id`, the row's id)\nnames the same single run, because `request-render` starts one run per row.\n\nNever key the claim by a hash of the content (the composition and props). Two users who ask for the\nsame render produce two runs with one hash: the second start would find the first's claim and be\nrefused, or overwrite the first's `callback_url`, and that run would wait out its deadline and refund.\nIf you want identical renders to share their output, claim by run id and look the output up by hash\nas a separate step.\n\n## Keys stay out of the container (the recommended shape)\n\nThe render container is the part of your system that runs the most third-party code (ffmpeg,\nChromium, fonts and media from the user's props), so give it **no vxil key at all**. This is the\nshape the blueprint recommends, whatever runs the container:\n\n```\nvxil \u2500\u2500start\u2500\u2500\u25B6 coordinator \u2500\u2500launch\u2500\u2500\u25B6 container\n (render_url) \u2502 progress / failed \u2500\u2500\u2500\u2500\u2500\u2500\u25B6 callback_url (keyless)\n \u25B2 output bytes \u2500\u2500\u2500\u2500\u2500\u2518\n \u2502\n \u2514\u2500 mints the upload URL with ITS files:write key, PUTs the bytes,\n completes the object, POSTs \"completed\" + output_file \u2500\u2500\u25B6 callback_url\n```\n\n- **The files feature is on.** The blueprint's config enables it (`files: { enabled: true }`) and\n declares `output_file` as a `file` field; without it every upload-url call is refused and the\n render waits out its deadline. Its per-object ceiling, `maxObjectBytes`, is 100 MB by default,\n enough for the buffered coordinator below; raise it for long or high-bitrate renders.\n- **The coordinator** is your `render_url`: a small endpoint in your own account \u2014 an edge worker\n with a per-render lock, or a route on any thin server. It is the only piece that holds a vxil key,\n and that key holds only **`files:write`** (`vxil keys mint --name render-uploads --scopes files:write`).\n- **The container** gets the job, the keyless `callback_url` (for `processing` pings and for\n `failed`), an `output_url` on the coordinator and an **output ticket** for it: a token signed for\n that one render and useless after its deadline, sent on the `Authorization` header (never in the\n URL, where access logs would keep it).\n- **The upload happens once the size is known.** The container POSTs the finished file to its\n ticket. The coordinator reads it, mints the files upload URL for exactly that size\n (the quota pre-check uses it), PUTs the bytes, completes the object, and only then posts\n `completed` with the object id as `output_file`. A crash anywhere before that leaves the render\n open, and it refunds at the deadline like any other.\n\n```ts\n// coordinator.ts \u2014 your render_url. A standard fetch handler (an edge worker, or a Node 18+ adapter).\n// Env: RENDER_TOKEN (= the vxil secret render_token), TICKET_SECRET (a long random string),\n// VXIL_FILES_KEY (an API key holding ONLY files:write), VXIL_BASE (https://api.vxil.com).\nimport { postCallback } from './callbacks';\n\ntype Start = {\n render_id: string; user_id: string; composition: string; props?: Record<string, unknown>;\n payload: { generation_id: string; deadline_at: string }; callback_url: string;\n};\ntype Ticket = { run_id: string; render_id: string; user_id: string; callback_url: string; exp: number };\ntype Env = { RENDER_TOKEN: string; TICKET_SECRET: string; VXIL_FILES_KEY: string; VXIL_BASE: string };\n\nconst enc = new TextEncoder();\nconst b64u = (b: ArrayBuffer | Uint8Array) =>\n btoa(String.fromCharCode(...new Uint8Array(b))).replace(/\\+/g, '-').replace(/\\//g, '_').replace(/=+$/, '');\nconst unb64u = (s: string) => Uint8Array.from(atob(s.replace(/-/g, '+').replace(/_/g, '/')), (c) => c.charCodeAt(0));\nconst hmacKey = (secret: string) =>\n crypto.subtle.importKey('raw', enc.encode(secret), { name: 'HMAC', hash: 'SHA-256' }, false, ['sign', 'verify']);\nasync function sealTicket(env: Env, t: Ticket): Promise<string> {\n const body = b64u(enc.encode(JSON.stringify(t)));\n return `${body}.${b64u(await crypto.subtle.sign('HMAC', await hmacKey(env.TICKET_SECRET), enc.encode(body)))}`;\n}\nasync function openTicket(env: Env, raw: string): Promise<Ticket | null> {\n const [body, sig] = raw.split('.');\n if (!body || !sig) return null;\n try {\n // crypto.subtle.verify compares in constant time (a `!==` on the signature would not)\n if (!(await crypto.subtle.verify('HMAC', await hmacKey(env.TICKET_SECRET), unb64u(sig), enc.encode(body)))) return null;\n const t = JSON.parse(new TextDecoder().decode(unb64u(body))) as Ticket;\n return t.exp > Date.now() ? t : null;\n } catch {\n return null; // not base64url / not JSON\n }\n}\n/** The output's file extension, from the Content-Type the container sends. */\nconst EXT: Record<string, string> = {\n 'video/mp4': 'mp4', 'video/webm': 'webm', 'image/gif': 'gif', 'image/png': 'png', 'image/jpeg': 'jpg', 'application/pdf': 'pdf',\n};\n/** Sent with a 503: when to send the request again. */\nconst AGAIN_IN_30S = { 'retry-after': '30' };\nconst vxil = (env: Env, path: string, body?: unknown) => fetch(`${env.VXIL_BASE}${path}`, {\n method: 'POST',\n headers: { authorization: `Bearer ${env.VXIL_FILES_KEY}`, 'content-type': 'application/json' },\n ...(body ? { body: JSON.stringify(body) } : {}),\n});\n\nexport default {\n async fetch(req: Request, env: Env): Promise<Response> {\n const url = new URL(req.url);\n\n // 1. the start: check the token, launch the container, answer inside 20 s\n if (req.method === 'POST' && url.pathname === '/start') {\n if (req.headers.get('authorization') !== `Bearer ${env.RENDER_TOKEN}`) return new Response('unauthorized', { status: 401 });\n const s = (await req.json()) as Start;\n // a run started before request-render sent user_id (an older copy of this\n // blueprint): a server-mode upload must name its user, so refuse the start \u2014\n // a 4xx ends that run and refunds it, instead of a 422 at upload time\n if (!s.user_id) return new Response('start body has no user_id: redeploy request-render', { status: 400 });\n // one run, one render: claim the work under the RUN id before launching anything. The claim\n // says `launching` (and expires after a minute) until the launch succeeds, then `launched`.\n // A start vxil re-sends (a lost answer) is acknowledged with 202 only once the work is\n // `launched`; while another copy is still launching it gets 503, so vxil keeps retrying\n // instead of treating a launch that may yet fail as accepted.\n // Never claim by a content hash: two users' identical renders are two runs.\n const runId = req.headers.get('x-vxil-run-id') ?? s.payload.generation_id;\n const claimKey = `start:${runId}`;\n if (!(await store.claim(claimKey, 'launching', 60_000))) {\n if ((await store.get(claimKey)) === 'launched') return Response.json({ accepted: true, duplicate: true }, { status: 202 });\n return new Response('this render is still being launched', { status: 503, headers: AGAIN_IN_30S });\n }\n const ticket = await sealTicket(env, {\n run_id: runId, render_id: s.render_id, user_id: s.user_id, callback_url: s.callback_url,\n exp: Date.parse(s.payload.deadline_at), // useless once vxil stops waiting\n });\n try {\n await launchContainer({ // YOUR container platform's API: it must QUEUE\n run_id: runId, render_id: s.render_id, // the job and return at once \u2014 never wait.\n // Pass run_id as its idempotency key if it has one\n composition: s.composition, props: s.props ?? {},// here for a cold start\n deadline_at: s.payload.deadline_at,\n callback_url: s.callback_url, // for processing pings and `failed`\n output_url: `${url.origin}/output`, // POST the file here, with\n output_ticket: ticket, // Authorization: Bearer <output_ticket>\n });\n } catch {\n await store.release(claimKey); // not launched: let vxil's retry try again\n return new Response('cannot take the render right now', { status: 503, headers: AGAIN_IN_30S });\n }\n await store.put(claimKey, 'launched'); // no expiry: every later copy is a duplicate\n return Response.json({ accepted: true }, { status: 202 });\n }\n\n // 2. the output: the container POSTs the finished file here (again, on any non-2xx answer)\n if (req.method === 'POST' && url.pathname === '/output') {\n const t = await openTicket(env, (req.headers.get('authorization') ?? '').replace(/^Bearer /, ''));\n if (!t) return new Response('bad or expired ticket', { status: 403 });\n const contentType = (req.headers.get('content-type') ?? '').split(';')[0]!.trim().toLowerCase();\n const ext = EXT[contentType];\n if (!ext) return new Response(`send the output's Content-Type (one of: ${Object.keys(EXT).join(', ')})`, { status: 415 });\n // a re-sent output after the upload already happened: skip straight to the settle\n let object_id = await store.get(`output:${t.run_id}`);\n if (!object_id) {\n // buffered: fine for outputs of tens of MB \u2014 see \"Very large outputs\" below\n const bytes = await req.arrayBuffer();\n const size = bytes.byteLength;\n if (size === 0) return new Response('empty output', { status: 400 });\n\n const minted = await vxil(env, '/v1/files/upload-url', {\n user_id: t.user_id, filename: `${t.render_id}.${ext}`, content_type: contentType, size_bytes: size,\n });\n if (!minted.ok) return new Response(`upload-url ${minted.status}`, { status: 502 }); // the container sends the file again\n const m = ((await minted.json()) as { data: { object_id: string; upload_url: string } }).data;\n const put = await fetch(m.upload_url, { method: 'PUT', headers: { 'content-type': contentType }, body: bytes });\n if (!put.ok) return new Response(`upload ${put.status}`, { status: 502 });\n const done = await vxil(env, `/v1/files/${encodeURIComponent(m.object_id)}/complete`);\n if (!done.ok) return new Response(`complete ${done.status}`, { status: 502 });\n object_id = m.object_id;\n await store.put(`output:${t.run_id}`, object_id);\n }\n\n // 3. settle the render on the keyless callback: a MUST-ARRIVE post. Retry here for up to a\n // minute (never past the deadline); if it still did not land, answer 503 and the container\n // sends the output again \u2014 the note above turns that into one more settle attempt.\n try {\n await postCallback(t.callback_url,\n { status: 'completed', output_file: object_id, progress: 100, stage: 'done' },\n Math.min(t.exp, Date.now() + 60_000));\n } catch (e) {\n return new Response(`callback: ${String(e)}`, { status: 503, headers: AGAIN_IN_30S });\n }\n return Response.json({ object_id });\n }\n return new Response('not found', { status: 404 });\n },\n};\n\ndeclare function launchContainer(job: Record<string, unknown>): Promise<void>; // your container platform's API\n/** Your coordinator's own small store: a key-value namespace, a table, a per-key lock. `claim` is\n * an insert-if-absent (of `value`, expiring after `ttlMs`) that answers true for the FIRST caller\n * only; `put` writes without an expiry. */\ndeclare const store: {\n claim(key: string, value: string, ttlMs: number): Promise<boolean>; release(key: string): Promise<void>;\n get(key: string): Promise<string | null>; put(key: string, value: string): Promise<void>;\n};\n```\n\nThe container's side is three kinds of HTTP call and no vxil key: `processing` pings to\n`callback_url`, the file to `output_url` (with `Authorization: Bearer <output_ticket>` and the\nfile's `Content-Type`), and `{\"status\": \"failed\", \u2026}` to `callback_url` if it gives up. The\ncontainer sends its pings through `progressReporter` and its `failed` through `postCallback`\n([callbacks.ts](#callback-limits-progress-is-best-effort-the-final-post-must-arrive)), and it sends\nthe output again on any non-`2xx` answer until `deadline_at`. Five notes on the shape:\n\n- **A doubled start launches one container.** The coordinator claims `start:<run id>` as\n `launching` before it launches and marks it `launched` once the launch succeeded. A start vxil\n re-sends is answered `202` and launches nothing only when the claim says `launched`; while the\n first copy is still launching, the re-sent one is answered `503`, so vxil keeps the start alive.\n (A `202` there would end vxil's retries, and if the first launch then failed nothing would run\n and the render would refund at its deadline.) When the launch fails, the coordinator releases\n the claim and answers `503` at once, and vxil sends the start again later. A coordinator that\n dies mid-launch leaves a `launching` claim that expires after a minute, so a later copy launches;\n pass the run id as your container platform's idempotency key, where it has one, so that copy\n cannot start a second container if the first launch did go through.\n- **A retried output post** (the container saw a network error after the coordinator had uploaded)\n finds `output:<run id>` in the coordinator's store, skips the upload and only sends the\n `completed` post again. A repeated `completed` is answered with the settled state and changes\n nothing, so the final post can be retried as often as it takes.\n- **Very large outputs.** The coordinator above holds the whole file in memory, and passing gigabytes\n through it costs its bandwidth too. For big files, have the container report the size first\n (`POST /output-url` with its ticket on the `Authorization` header and `{ size_bytes }`) and let the coordinator answer with the\n presigned upload URL it minted; the container PUTs straight to it, and the coordinator completes\n the object and posts `completed` when the container says it is done. The container still holds no\n key: a presigned URL is good for one object for a few minutes.\n- **Upgrading an earlier copy of this blueprint.** Renders started before `request-render` put\n `user_id` in the start body have none, and a server-mode upload must name its user. The\n coordinator refuses such a start with a `400`, which ends that run and refunds it at once;\n redeploy `request-render` (`vxil push`) before you point `render_url` at the coordinator.\n- **Serving it.** The row now holds a files object id. Read it back with a signed download URL, or\n publish it from a settle function with a server key (`vx.files.publish(object_id)`) for a stable\n public URL served from the edge cache.\n\nOptions A and B below show the same contract with the runtime posting `completed` itself (an\n`output_url` in your own bucket). Either can adopt the coordinator: point `render_url` at it, and have\nstep 1 trigger the Trigger.dev task or spawn the Modal function.\n\n## Option A \u2014 Trigger.dev (v4)\n\nTrigger.dev runs the render as a task on a machine you pick, with ffmpeg or Chromium baked into the\nimage, retries, and its own run dashboard. Two pieces: a **relay endpoint** that turns vxil's start\nrequest into a Trigger.dev trigger, and the **task**.\n\n**Why a relay, and not `render_url` pointed straight at Trigger.dev's trigger API?** vxil sends\n`callback_url` beside `payload` at the top level of the body, and Trigger.dev's trigger API passes\nonly `payload` to the task \u2014 the task would never see where to report. The relay is ~30 lines and is\nalso where your `render_token` is checked. Host it anywhere that serves https: a serverless function on\nyour web host, a small edge worker, a route in your existing API.\n\n```ts\n// relay.ts \u2014 your render_url. Standard fetch handler (edge worker / serverless function / Node 18+ adapter).\n// Env: RENDER_TOKEN (the same value as the vxil secret render_token), TRIGGER_SECRET_KEY (tr_prod_\u2026 / tr_dev_\u2026).\ntype Start = {\n render_id: string; composition: string; props?: Record<string, unknown>;\n payload: { generation_id: string; correlation_id?: string; deadline_at: string }; callback_url: string;\n};\n\nexport default {\n async fetch(req: Request, env: { RENDER_TOKEN: string; TRIGGER_SECRET_KEY: string }): Promise<Response> {\n if (req.method !== 'POST') return new Response('method not allowed', { status: 405 });\n if (req.headers.get('authorization') !== `Bearer ${env.RENDER_TOKEN}`) {\n return new Response('unauthorized', { status: 401 }); // a 4xx ends the vxil run (and refunds)\n }\n const s = (await req.json()) as Start;\n const res = await fetch('https://api.trigger.dev/api/v1/tasks/render-video/trigger', {\n method: 'POST',\n headers: { authorization: `Bearer ${env.TRIGGER_SECRET_KEY}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n payload: {\n render_id: s.render_id, composition: s.composition, props: s.props ?? {},\n callback_url: s.callback_url, deadline_at: s.payload.deadline_at,\n },\n options: {\n // a start vxil re-sends (a lost answer) triggers the SAME Trigger.dev run\n idempotencyKey: `render:${s.render_id}`,\n // a render still queued after 3 minutes is dropped (it never runs, and no\n // onFailure fires \u2014 vxil refunds it at its deadline). Part of the budget below.\n ttl: '3m',\n tags: [`render_${s.render_id}`],\n },\n }),\n });\n if (res.ok) return Response.json({ accepted: true }, { status: 202 });\n // Trigger.dev busy or down: answer 503 and vxil tries the start again; anything else ends the run\n return new Response(`trigger.dev ${res.status}`, { status: res.status === 429 || res.status >= 500 ? 503 : 400 });\n },\n};\n```\n\n```ts\n// trigger.config.ts \u2014 ffmpeg and Chromium in the task image\nimport { defineConfig } from '@trigger.dev/sdk';\nimport { ffmpeg } from '@trigger.dev/build/extensions/core';\nimport { puppeteer } from '@trigger.dev/build/extensions/puppeteer';\n\nexport default defineConfig({\n project: '<your project ref>',\n dirs: ['./trigger'],\n maxDuration: 600, // CPU seconds PER ATTEMPT \u2014 the task sets its own; see the budget below\n build: { extensions: [ffmpeg(), puppeteer()] }, // puppeteer also needs PUPPETEER_EXECUTABLE_PATH set in the Trigger.dev env\n});\n```\n\n```ts\n// trigger/render-video.ts \u2014 the task: render, upload, report to vxil\nimport { task, metadata, logger } from '@trigger.dev/sdk';\nimport { postCallback, progressReporter } from '../callbacks'; // the helpers above\n\ntype Payload = {\n render_id: string; composition: string; props: Record<string, unknown>;\n callback_url: string; deadline_at: string; // when vxil stops waiting (ISO)\n};\n\n/** The longest one attempt takes, wall clock, with margin. An attempt that cannot\n * finish before deadline_at does not start: it tells vxil, so the credits come back now. */\nconst ATTEMPT_WALL_MS = 12 * 60_000;\n\n/** completed / failed: retried on 429 / 5xx / network errors (Retry-After honoured) until the\n * deadline \u2014 a repeated final post is a no-op on vxil's side, so retrying is always safe. */\nconst report = (p: Payload, body: Record<string, unknown>) =>\n postCallback(p.callback_url, body, Date.parse(p.deadline_at));\n\nexport const renderVideo = task({\n id: 'render-video',\n machine: 'large-1x', // 4 vCPU / 8 GB \u2014 size to your renders\n maxDuration: 600, // CPU seconds per attempt (10 min) \u2014 not wall time\n retry: { maxAttempts: 2, minTimeoutInMs: 5_000, maxTimeoutInMs: 30_000 },\n run: async (p: Payload) => {\n if (Date.now() + ATTEMPT_WALL_MS > Date.parse(p.deadline_at)) {\n // too late to finish inside vxil's deadline: refund now, and do not retry\n await report(p, { status: 'failed', error: 'deadline', hint: 'no time left for another attempt' });\n return { skipped: 'deadline' };\n }\n metadata.set('stage', 'rendering'); // Trigger.dev's own run view\n const progress = progressReporter(p.callback_url); // best-effort, at most one ping per 5 s\n await progress({ stage: 'rendering', progress: 0 });\n\n // \u2026your render: drive Chromium for frames, run ffmpeg \u2014 call progress({ progress, stage })\n // as often as you like (it coalesces) \u2014 and write the file\n // to YOUR bucket keyed by render_id (so a retried attempt overwrites, not duplicates)\u2026\n const outputUrl = `https://cdn.example.com/renders/${p.render_id}.mp4`;\n const durationS = 31.2;\n logger.info('rendered', { render_id: p.render_id, outputUrl });\n if (Date.now() > Date.parse(p.deadline_at)) {\n // vxil has already failed and refunded this render: the post below is answered\n // with that settled state. Your sizing is off \u2014 widen the budget below.\n logger.warn('finished after the vxil deadline', { render_id: p.render_id });\n }\n\n await report(p, {\n status: 'completed', output_url: outputUrl, duration_s: durationS, progress: 100, stage: 'done',\n });\n return { output_url: outputUrl };\n },\n // after the last attempt THROWS: tell vxil, so the credits come back now, not at the\n // deadline. Not called when an attempt exceeds maxDuration or the run expires on its\n // ttl \u2014 those refund only at vxil's deadline.\n onFailure: async ({ payload, error }) => {\n await report(payload, {\n status: 'failed', error: 'render_failed', hint: String(error instanceof Error ? error.message : error).slice(0, 200),\n });\n },\n});\n```\n\n**Budget the wall clock.** Trigger.dev's limits and vxil's deadline are separate clocks, and only\nvxil's refunds. Size them so a render always ends \u2014 `completed` or `failed` \u2014 before vxil's deadline:\n\n```\nttl + maxAttempts \xD7 (longest attempt, wall clock) + retry backoff < timeout.after_ms\n3 min + 2 \xD7 12 min + \u2264 1 min = 28 min < 30 min\n```\n\n`maxDuration` counts **CPU time per attempt**, not wall time across the run, so it does not bound\nthe sum: the `deadline_at` check at the start of each attempt does. If your renders need more, raise\n`RENDER_DEADLINE_MS` in `request-render` (up to `generation.maxTimeoutMs`, one hour) and resize the\nrest to fit.\n\n**Be honest with yourself about three things before you ship on Trigger.dev Cloud:**\n\n- **Data residency.** Trigger.dev Cloud keeps its operational and log data \u2014 including each run's\n payload \u2014 in the US (us-east-1), even when the machines run elsewhere. Your render props and the\n `callback_url` pass through it. If that rules it out, self-host Trigger.dev for your own app, or use\n option B.\n- **The callback URL is a credential for one render.** It appears in Trigger.dev's run payload and\n dashboard. It can settle only that render, and stops mattering once the render is settled.\n- **Two clocks, and `onFailure` is not a guarantee.** Trigger.dev calls `onFailure` only after the last\n attempt throws. A run that exceeds `maxDuration`, or expires on its `ttl` before it starts, ends\n without it \u2014 vxil refunds those at its deadline, not sooner. Keep the budget above, and keep the\n `deadline_at` check, so a late attempt refunds early instead of finishing after the refund.\n\n## Option B \u2014 your own container runtime (Modal, Fly, a container on your own cloud account)\n\nSame contract, no relay: the endpoint you deploy **is** `render_url`. It must answer within 20 seconds,\nso it only checks the token, hands the job to a background worker and answers `2xx`; the worker renders\nand posts back. On Modal:\n\n```python\n# render_app.py \u2014 `modal deploy render_app.py`; render_url = the endpoint's https URL\nimport http.client, json, os, time, urllib.error, urllib.request\nfrom datetime import datetime, timedelta, timezone\nimport modal\nfrom fastapi import HTTPException, Request # also `pip install fastapi` where you run `modal deploy`\n\nimage = (modal.Image.debian_slim()\n .apt_install(\"ffmpeg\", \"chromium\")\n .pip_install(\"fastapi[standard]\"))\napp = modal.App(\"render-farm\", image=image)\nsecrets = [modal.Secret.from_name(\"render-farm\")] # RENDER_TOKEN\n\ndef _post(callback_url: str, body: dict) -> None:\n req = urllib.request.Request(callback_url, data=json.dumps(body).encode(),\n headers={\"content-type\": \"application/json\"}, method=\"POST\")\n with urllib.request.urlopen(req, timeout=30) as res:\n res.read()\n\ndef report(callback_url: str, body: dict, deadline: datetime) -> None:\n \"\"\"completed / failed: MUST arrive. Retries 429, 5xx and network errors (honouring\n Retry-After) until the deadline; a repeated final post changes nothing on vxil's side.\"\"\"\n attempt = 0\n while True:\n wait = min(2 ** attempt, 30)\n attempt += 1\n try:\n _post(callback_url, body)\n return\n except urllib.error.HTTPError as e:\n if e.code != 429 and e.code < 500:\n raise # another 4xx: the URL or the body is wrong\n ra = e.headers.get(\"retry-after\")\n wait = int(ra) if ra and ra.isdigit() else wait\n except (OSError, http.client.HTTPException):\n pass # no answer, or the connection dropped mid-answer\n # (URLError, a reset, RemoteDisconnected, IncompleteRead,\n # a timeout): send it again\n if datetime.now(timezone.utc) + timedelta(seconds=wait) > deadline:\n raise RuntimeError(\"callback not delivered before the deadline\")\n time.sleep(wait)\n\n_last_ping = {}\ndef ping(callback_url: str, body: dict) -> None:\n \"\"\"processing: best-effort, at most one every 5 s; a refused or failed ping is dropped.\"\"\"\n now = time.monotonic()\n if now - _last_ping.get(callback_url, 0.0) < 5:\n return\n _last_ping[callback_url] = now\n try:\n _post(callback_url, {\"status\": \"processing\", **body})\n except Exception:\n pass\n\nATTEMPT_WALL_S = 25 * 60 # = the timeout below; a call cut off there may never reach its except\n\n@app.function(cpu=4, memory=8192, timeout=ATTEMPT_WALL_S, secrets=secrets)\ndef render(job: dict) -> None:\n cb = job[\"callback_url\"]\n deadline = datetime.fromisoformat(job[\"payload\"][\"deadline_at\"].replace(\"Z\", \"+00:00\"))\n if datetime.now(timezone.utc) + timedelta(seconds=ATTEMPT_WALL_S) > deadline:\n # queued too long to finish before vxil stops waiting: refund now\n report(cb, {\"status\": \"failed\", \"error\": \"deadline\", \"hint\": \"started too late to finish\"}, deadline)\n return\n try:\n ping(cb, {\"stage\": \"rendering\", \"progress\": 0})\n # \u2026render with ffmpeg / chromium (ping(cb, {...}) as often as you like),\n # upload to YOUR bucket keyed by job[\"render_id\"]\u2026\n output_url = f\"https://cdn.example.com/renders/{job['render_id']}.mp4\"\n except Exception as e: # the RENDER failed: tell vxil now, so the credits come back before the deadline\n report(cb, {\"status\": \"failed\", \"error\": \"render_failed\", \"hint\": str(e)[:200]}, deadline)\n raise\n # the output exists: only `completed` may follow. Outside the try on purpose, so a\n # trouble delivering it can never turn into a `failed` post that refunds a finished render.\n report(cb, {\"status\": \"completed\", \"output_url\": output_url, \"duration_s\": 31.2,\n \"progress\": 100, \"stage\": \"done\"}, deadline)\n\n@app.function(secrets=secrets)\n@modal.fastapi_endpoint(method=\"POST\")\nasync def start(request: Request):\n if request.headers.get(\"authorization\") != f\"Bearer {os.environ['RENDER_TOKEN']}\":\n raise HTTPException(status_code=401, detail=\"unauthorized\") # a 4xx ends the vxil run\n job = await request.json()\n await render.spawn.aio(job) # queued; returns at once, well inside the 20-second window\n return {\"accepted\": True}\n```\n\n`spawn` queues the call and returns immediately, so the endpoint answers in well under a second. A\nstart can arrive twice (a lost answer is retried), so dedupe on the run id\n(`request.headers[\"x-vxil-run-id\"]`; never a hash of the props): keep the run ids you have spawned in\na `modal.Dict`, or make the render overwrite the same output key.\n\nAnything else that can (1) answer an https POST in under 20 s, (2) run the work in the background and\n(3) POST JSON to a URL fits the same contract: a Fly Machine started per job, a container service with\na queue in front, your own GPU box. vxil does not care what runs the render \u2014 only that the start is\nacknowledged quickly and the callback eventually comes.\n\n## Run it\n\nGive a user some credits from your server (or sell `render_pack_100` through your payments provider):\n\n```bash\ncurl -s -X POST \"https://api.vxil.com/v1/payments/credits/grant\" \\\n -H \"authorization: Bearer $KEY\" -H 'content-type: application/json' \\\n -H 'idempotency-key: welcome-u1' \\\n -d '{\"user_id\":\"<the user id>\",\"credit_type\":\"render_credits\",\"amount\":25,\"source\":\"welcome\"}'\n```\n\nStart a render **with the user's session** (end-user mode \u2014 the held credits are forced onto that\nuser):\n\n```ts\nimport { Vxil } from '@vxil/sdk';\n\n// after `vxil gen`, vx.fn['request-render'] is typed from the function's declared signature\nconst vx = new Vxil({ apiKey: process.env.VXIL_PUBLISHABLE_KEY!, endUserToken: process.env.USER_SESSION! });\nconst started = await vx.fn['request-render']({\n composition: 'promo-30s', request_key: 'promo-1', props: { headline: 'Spring sale' },\n});\n// \u2192 { item_id, run_id, credits: 5 }\n// (or { duplicate: true, request_key, item_id, run_id, status } on a retry)\n```\n\nThen read the row \u2014 or subscribe to its changes \u2014 until `status` is `completed`:\n\n```ts\nif ('item_id' in started) {\n const row = await vx.from('renders').get(started.item_id);\n // row.status \u2192 'completed', row.output_url \u2192 'https://cdn.example.com/renders/\u2026.mp4'\n}\n```\n\n**Try it before you have a runtime.** Point `render_url` at any https endpoint that answers `2xx`\n(a request-bin works) and play the runtime yourself: copy `callback_url` from the request it received,\nthen\n\n```bash\ncurl -s -X POST \"$CALLBACK_URL\" -H 'content-type: application/json' \\\n -d '{\"status\":\"completed\",\"output_url\":\"https://cdn.example.com/x.mp4\",\"duration_s\":12.5,\"progress\":100,\"stage\":\"done\"}'\n# \u2192 { \"data\": { \"run_id\": \"run_\u2026\", \"generation_status\": \"completed\" } } \u2014 and the row says so\n```\n\n## How it fails, and what the user sees\n\n| what happened | the run | the row | the credits |\n|---|---|---|---|\n| runtime posted `completed` | `completed` | `status: completed` + every key it sent | committed |\n| runtime posted `failed` | `failed` | `status: failed`, `error` = its code + hint (written by `notify-ready`) | refunded |\n| runtime never called back | `failed` (`GenerationExpired`) at the deadline | `failed`, `error: GenerationExpired: the render did not finish before its deadline` | refunded |\n| runtime finished after the deadline | `failed` (`GenerationExpired`) \u2014 exact to the second; the late `completed` is answered `failed` / `expired: true` | `failed` | refunded (your compute was spent \u2014 budget the clocks) |\n| the final post was answered `429` (or `5xx`, or got no answer) | still open: nothing is settled until a post lands | unchanged | still held \u2014 `postCallback` sends it again after `Retry-After`; give up only at `deadline_at`, when the run refunds anyway |\n| a progress ping was answered `429` | unchanged | the previous progress stays | unchanged \u2014 drop the ping; the next one carries the newer state |\n| your endpoint answered `5xx` / timed out | start retried with backoff; terminal after the attempts | `processing` \u2192 `failed`, `error: RetryableHttp: the render endpoint kept failing to accept the render (retries exhausted)` (or `NetworkError: \u2026` when it could not be reached) | held until then, then refunded |\n| your endpoint answered another `4xx` (a bad token) | `failed` at once | `failed` | refunded |\n| the user had too few credits (at `request-render`) | ended at once (`ReserveInsufficient`), never started | `failed`, `error: insufficient_credits`; that `request_key` is spent; the caller got the `402`, no message | nothing held |\n| too many renders in flight, payments briefly unreachable, or a jobs-side fault | not created yet (the caller gets `429` / `503` with `queued: true` and `retry_after`, or `502` with `queued: true`) | `pending`, no run \u2014 **queued**: `redrive-pending` starts it when a slot frees up (or the app calls again with the **same** `request_key`) | held when it starts |\n| the user had too few credits when the re-driver started it | ended at once (`ReserveInsufficient`) | `failed`, `error: insufficient_credits`; the owner is told once (they last heard \"queued\") | nothing held |\n| still no free slot an hour later | never created | `failed`, `error: not_started: the render waited too long for a free slot` (written by `redrive-pending`); the owner is told once | nothing held |\n| the re-driver's start was refused (`400` / `401` / `403` / `422`: a non-https or private `render_url`, a wrong or revoked `vxil_jobs_key`) | not created | still `pending` \u2014 the tick stops and reports it (`stopped: { status, code }` in the function's logs); fix the setup and the next tick carries on; the one-hour bound still applies | nothing held |\n\n## The backlog: `maxConcurrent`, `429` and the re-driver\n\n`generation.maxConcurrent` is how many generation runs this project may have **open** at once: 20 by\ndefault, settable up to 200 in the jobs config. This blueprint sets 20; raise it to what your plan\nand your runtime can carry:\n\n```ts\njobs: { enabled: true, generation: { maxConcurrent: 50 /* 1\u2013200, default 20 */ } },\n```\n\nAt the cap a new start is refused with `429` and `Retry-After: 5`. **No run is created and nothing\nis held yet.** The same holds when the credits cannot be held because payments is briefly\nunreachable: the start is refused with `503 payments_unavailable` (nothing started, nothing held;\nit names a 15-second wait), `request-render` answers `503` with `queued: true`, and the re-driver starts\nthe row on a later tick with the same key (a fresh run). Never a free render: that only happens if\nyou opt in with `reserve_credits.on_unavailable: 'proceed'`. `request-render` passes the `429` on to the app with `queued: true` (and the\n`item_id` and `request_key`), and leaves the row `pending` with no `run_id`. That render is\n**queued, not refused**: the re-driver will start it, and hold its credits then, up to an hour\nlater. So the app must treat a `429` with `queued: true` as \"queued\" \u2014 show it, watch the row \u2014 and\nmust **never retry it with a new `request_key`**: that is a second render, and both are charged. A\nretry with the **same** `request_key` is always safe. Rows like that are the backlog, and two\nthings drain it:\n\n1. **`redrive-pending`, every minute.** It reads the oldest `pending` rows with no run that are at\n least 30 s old (`{ status: 'pending', created_at: { $lt: \u2026 }, run_id: null }`, sorted by\n `created_at`, 20 per tick; `status` and `created_at` are index slots, so the read stays cheap at\n any size) and starts each one with the same descriptor and the **same idempotency key** as\n `request-render`, so a row the app is re-driving at the same moment still gets one run. It\n **stops at the first `429`** (or `503 payments_unavailable`), since the rest of the batch would\n get the same answer, and the next tick carries on. Each try is recorded on the row (`redrive_attempts`, `redriven_at`). A `402`\n fails the row with `insufficient_credits`. A `400`, `401`, `403` or `422` also **stops** the tick\n and fails nothing: every re-driven row sends the same descriptor, so a refusal means the setup is\n wrong (a non-https or private `render_url`, a revoked key), not the row; the tick reports it as\n `stopped: { status, code }`. The only thing that fails a waiting row is age: a row still waiting\n after **one hour** is failed with `not_started: the render waited too long for a free slot`.\n Nothing was ever held, so nothing is refunded. In both the `402` and the one-hour case the owner\n is told once (`Idempotency-Key: render-not-started:<item_id>`), because the last thing they heard\n was \"queued\". (The function's scopes include `notifications:send` for that.)\n2. **The app**, calling `request-render` again with the same `request_key`, if it wants the render\n started sooner than the next tick.\n\n`overlap: 'skip'` keeps a slow tick from being doubled by the next one. On the Free plan, run it\nevery 15 minutes (see the plan note at the top).\n\n**Why it has its own key.** A credit hold placed by a function must name the signed-in user the\nfunction acts for, and a cron tick has none. So `redrive-pending` makes the start with\n`vxil_jobs_key`, an API key holding only `jobs:write`, the way your trusted server would. That is\nmore than \"hold credits and start runs\": `jobs:write` also lets the key cancel or replay any run of\nthis project, enqueue any job, and create or change schedules and flow rules, and it can hold\ncredits against any of your users and start runs against any public https endpoint. Treat it as a\nserver key: keep it only in this function's secrets, never in an app, and rotate it like any\nserver key. The re-driver never reads the\nprice from the row (a signed-in user can edit their own row): the hold is the function's fixed\n`RENDER_CREDITS`, and a row whose `render_key` is not `owner:request_key` is failed, not started.\n(A hand-written row like that which also breaks the `render_key_shape` hook cannot be written at\nall, so the tick counts it as `unwritable` and it keeps one of the 20 slots: fix or delete it by\nhand.)\n\n## A contract test for your runtime\n\nEvery key your runtime posts in the `completed` body is written onto the row, so every key must be a\ndeclared field of `renders`. Keep that true in your own CI with a test beside your runtime's code. It\nfails the day someone adds a key to the completion body and forgets the field:\n\n```ts\n// render-contract.test.ts \u2014 vitest, in YOUR repo (the one holding vxil.config.ts)\nimport { describe, expect, it } from 'vitest';\nimport config from './vxil.config';\n// the bodies your runtime (or coordinator) really posts: import the builders from that code,\n// so the test follows it instead of a hand-copied list\nimport { completedBody, progressBody } from './runtime/callback-bodies';\n\ndescribe('render callbacks', () => {\n const declared = new Set(Object.keys(config.cms!.collections!.renders!.fields!));\n\n it('every key of the completion body is a declared renders field', () => {\n const body = completedBody({ objectId: 'obj_test', durationS: 1 });\n expect(Object.keys(body).filter((k) => !declared.has(k))).toEqual([]);\n });\n\n it('a progress ping carries only the keys the mirror keeps', () => {\n const ping = progressBody({ progress: 40, stage: 'encoding' });\n expect(Object.keys(ping).filter((k) => !['status', 'progress', 'stage', 'message'].includes(k))).toEqual([]);\n });\n});\n```\n\nThe platform also has a fallback for the day that test is missing. When the row refuses a completion\nmirror because of an undeclared key, vxil writes the status alone, so the row still says\n`completed`, and the run reports what it dropped (`GET /v1/jobs/runs/{run_id}` \u2192 `mirror_error`,\nwith `fields_dropped`). The app no longer hangs on `processing`, but the dropped keys are not on the\nrow. The test is what keeps them there.\n\n## The bounds to design against\n\n- **Deadline**: the run's timeout is clamped to `generation.maxTimeoutMs` \u2014 one hour at most. A render\n that can take longer should be split (the next step starts from `notify-ready`), or tracked on your own\n row without a platform-held reserve.\n- **In flight**: `generation.maxConcurrent` renders at once (20 here and by default; up to 200). Over\n it, `request-render` answers `429` and the row waits for `redrive-pending`, or for a retry with the\n same `request_key` ([the backlog](#the-backlog-maxconcurrent-429-and-the-re-driver)).\n- **Holds**: one render's reserve is clamped to `generation.maxReserveCredits` (50 here), and the sum of\n all open holds is capped by `generation.maxOutstandingReserveCredits` (5,000 here).\n- **Callback**: at most 256 KiB per post, JSON, every key a declared field of `renders`. About 120\n posts a minute per render plus a small allowance kept for the final post (both best-effort, per\n serving machine); progress at most every 5 s, and the final post retried until it lands\n ([callback limits](#callback-limits-progress-is-best-effort-the-final-post-must-arrive)).\n\n## Your token, and where it lives\n\n`render_url` and `render_token` are function secrets; `request-render` reads them at invoke time and\nputs them on the generation run, which vxil stores with the run (your project only) for as long as the\njobs retention keeps it. The token is **never returned by a read**: `GET /v1/jobs/runs/{run_id}` (and\nthe dashboard and MCP reads built on it) shows the provider header names with every value as\n`[redacted]`. Rotate it in both places (`vxil secrets set functions/render_token` and your endpoint's\nenv); runs already started keep the token they were started with.\n\n## Evidence\n\n- **Read in vxil's code**: the start request body (`provider.body` + `payload` + `callback_url`), the\n 20-second start bound, the retry ladder, the status mirror writing every completion key onto the row,\n the hold committed on `completed` and released on `failed` / the deadline, and `error` / `hint`\n becoming `error_class` / `error_hint` on `job.generation.failed`.\n- **Read in Trigger.dev's and Modal's documentation, not executed from vxil**: the trigger endpoint\n `POST https://api.trigger.dev/api/v1/tasks/{taskId}/trigger` with `{ payload, options }` and the\n `idempotencyKey` / `ttl` / `tags` / `machine` options; `task({ id, machine, maxDuration, retry, run,\n onFailure })` and `retry` options, `maxDuration` being CPU time per attempt with no `onFailure` when\n it is exceeded, `metadata.set`, and the `ffmpeg()` / `puppeteer()` build extensions; Trigger.dev\n Cloud's US-hosted operational data; Modal's `@modal.fastapi_endpoint` and `.spawn()`. Check each\n vendor's current docs before you ship.\n",
18447
18566
  "functions": {
18448
- "notify-ready.ts": "// notify-ready.ts \u2014 job.generation.completed | failed \u2192 tell the render's owner\n// (a vxil function, webhook trigger on `job.generation.`).\n//\n// The status mirror has already written the row (status, and on completion\n// every key your runtime sent back). This function adds the two things a\n// mirror cannot: the failure cause on the row, and a message to the user.\n//\n// AT-LEAST-ONCE: an event can be delivered again. And because the trigger\n// declares `retry: { maxAttempts: 3 }`, a non-2xx answer from here goes back\n// to the jobs ladder for another attempt (without `retry` it would simply be\n// acknowledged). Every attempt carries the same event. The notification carries an\n// Idempotency-Key of one per RUN, so a redelivery sends nothing twice, and the\n// row patch writes the same value again.\n\nimport type { JobGenerationSettledEventPayload, WebhookFunctionEnvelope } from '@vxil/sdk';\n\ntype RenderRow = { owner?: string; composition?: string; output_url?: string; error?: string; redrive_attempts?: number };\n\n/** Platform error classes settle with no hint; these are the ones a render\n * can end with, in words for the row (the class stays first, for code). */\nconst PLATFORM_HINTS: Record<string, string> = {\n GenerationExpired: 'the render did not finish before its deadline',\n RetryableHttp: 'the render endpoint kept failing to accept the render (retries exhausted)',\n NetworkError: 'the render endpoint could not be reached (retries exhausted)',\n};\n\nfunction patchError(base: string, H: Record<string, string>, id: string, error: string): Promise<Response> {\n return fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(id)}`, {\n method: 'PATCH', headers: H, body: JSON.stringify({ data: { error } }),\n });\n}\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as WebhookFunctionEnvelope<JobGenerationSettledEventPayload>;\n const d = env.payload?.data;\n // job.generation.queued carries no status; a truncated event has no fields\n if (!d || 'truncated' in d || !('status' in d) || !d.generation_id) return Response.json({ skipped: true });\n const cms = env.scoped_jwts?.cms;\n const notifications = env.scoped_jwts?.notifications;\n if (!cms || !notifications) return Response.json({ error: 'missing cms/notifications scope' }, { status: 403 });\n const base = env.vxil_base;\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n\n const rowRes = await fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(d.generation_id)}`, { headers: H });\n // another job's generation event (not a render) \u2014 nothing to do\n if (rowRes.status === 404) return Response.json({ skipped: 'not a render' });\n if (!rowRes.ok) return Response.json({ error: `render read: ${rowRes.status}` }, { status: 502 }); // another attempt (header note)\n const row = ((await rowRes.json()) as { data: { data: RenderRow } }).data.data;\n if (!row.owner) return Response.json({ skipped: 'no owner' });\n\n if (d.status === 'failed') {\n // Not enough credits. A render request-render started itself already\n // answered the user (402) and wrote error: 'insufficient_credits' \u2014 keep\n // that code on the row (write it only if that write was lost) and send\n // nothing. A render the RE-DRIVER started is different: the user last\n // heard 429 \"queued\", so they are told \u2014 with the re-driver's own key\n // (render-not-started:<item_id>), so the two never both send.\n if (d.error_class === 'ReserveInsufficient') {\n if (!row.error) {\n const patched = await patchError(base, H, d.generation_id, 'insufficient_credits');\n if (!patched.ok) return Response.json({ error: `render patch: ${patched.status}` }, { status: 502 }); // another attempt (header note)\n }\n if (!(typeof row.redrive_attempts === 'number' && row.redrive_attempts > 0)) {\n return Response.json({ ok: true, notified: false });\n }\n return send(base, notifications, `render-not-started:${d.generation_id}`, row.owner, {\n subject: 'Your render could not start',\n paragraph: `\"${row.composition ?? 'Your render'}\" was queued, but there were not enough credits when its turn came. Nothing was charged \u2014 top up and try again.`,\n });\n }\n // the cause the run settled with: your runtime's `error` (+ `hint`), or\n // the platform's own class \u2014 which carries no hint, so a short human\n // one is added for the ones a user can meet (PLATFORM_HINTS)\n const hint = d.error_hint ?? (d.error_class ? PLATFORM_HINTS[d.error_class] : undefined);\n const cause = [d.error_class, hint].filter(Boolean).join(': ') || 'render failed';\n if (row.error !== cause) {\n const patched = await patchError(base, H, d.generation_id, cause.slice(0, 500));\n if (!patched.ok) return Response.json({ error: `render patch: ${patched.status}` }, { status: 502 }); // another attempt (header note)\n }\n }\n\n return send(base, notifications, `render-ready:${d.run_id}`, row.owner, d.status === 'completed'\n ? { subject: 'Your render is ready', paragraph: `\"${row.composition ?? 'Your render'}\" finished. Open the app to watch it.` }\n : { subject: 'Your render could not finish', paragraph: 'Nothing was charged \u2014 the credits are back on your balance. Try again in a minute.' });\n },\n};\n\n/** One transactional message; a non-2xx answer asks the ladder for another attempt. */\nasync function send(\n base: string, token: string, idempotencyKey: string, userId: string,\n data: { subject: string; paragraph: string },\n): Promise<Response> {\n const sent = await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json', 'idempotency-key': idempotencyKey },\n body: JSON.stringify({ user_id: userId, template: 'transactional', data }),\n });\n return sent.ok\n ? Response.json({ ok: true, notified: true })\n : Response.json({ error: `notifications send: ${sent.status}` }, { status: 502 }); // another attempt (header note)\n}\n",
18449
- "redrive-pending.ts": "// redrive-pending.ts \u2014 the BACKLOG RE-DRIVER (a vxil function, cron trigger,\n// every minute; `overlap: 'skip'`).\n//\n// cron-walk: drains-filter \u2014 every row it starts is PATCHed with its run_id (and\n// every row it gives up on to status 'failed'), which takes it out of the\n// `{ status: 'pending', run_id: null }` read; a 429 stops the tick early.\n//\n// Why it exists: at the generation concurrency cap (`generation.maxConcurrent`\n// open runs \u2014 20 in this blueprint, up to 200) request-render answers 429 and\n// leaves the row `pending` with NO run. Without this function only the caller\n// re-drives it (the same request_key again). With it, a backlog drains on its\n// own: each tick reads the oldest pending rows that have had no run for at\n// least REDRIVE_AFTER_MS, and starts each one with the SAME descriptor and the\n// SAME idempotency key (the row's render_key) request-render uses \u2014 so a row a\n// user is re-driving at the same moment still gets exactly one run.\n//\n// per row, by the jobs answer:\n// 202 (new run, or an open one handed back) \u2192 PATCH run_id (the status\n// mirror settles the row from here)\n// 202 deduplicated, generation_status 'failed' \u2192 PATCH run_id + status 'failed'\n// 402 (too few credits) \u2192 PATCH status 'failed',\n// error 'insufficient_credits',\n// and TELL the owner (they last\n// heard \"queued\", not \"refused\")\n// 429 (cap reached / too many credits held) \u2192 STOP the tick; the rest wait\n// for the next one (Retry-After\n// is reported)\n// 400 / 401 / 403 / 422 \u2192 STOP the tick, report it. Every\n// re-driven row sends the same\n// descriptor apart from its own\n// (already bounded) composition and\n// props, so a refusal is a SETUP\n// error \u2014 a non-https or private\n// render_url, a wrong or revoked\n// key, a config change \u2014 that\n// would fail every row the same\n// way. Nothing is failed for it:\n// fix the setup and the next tick\n// carries on.\n// 5xx / network \u2192 record the attempt, go on\n// A row still pending with no run after BACKLOG_MAX_AGE_MS is the ONLY thing\n// the re-driver gives up on: status 'failed', error 'not_started: \u2026' (nothing\n// was ever held for it), and the owner is told once\n// (Idempotency-Key render-not-started:<item_id>).\n// Every try is recorded on the row: redrive_attempts, redriven_at.\n//\n// A HAND-WRITTEN ROW that breaks render_key_shape (written before the hook\n// existed, or by a path that bypassed it) cannot be patched either \u2014 the\n// hook judges the merged row \u2014 so it would stay pending and take one of the\n// tick's slots for good. Fix or delete such rows by hand; the tick reports\n// them as `unwritable`.\n\n// THE KEY: a function-originated credit hold must name the user it acts for,\n// and a cron tick has no signed-in user \u2014 so the jobs call is made with\n// `vxil_jobs_key`, an API key of this same backend holding ONLY `jobs:write`\n// (a trusted server key may hold credits for any of your users; README \"The\n// backlog\"). The rows are read and written with the function's own scoped cms\n// token. The credits held are this blueprint's fixed price, never the row's\n// `credits` field \u2014 a signed-in user can edit their own row, so nothing the\n// re-driver charges or starts is read from a value they could have lowered.\n// The owner is told with the function's scoped notifications token.\n\nimport type { CronFunctionEnvelope } from '@vxil/sdk';\n\n/** What one render costs \u2014 the SAME number as request-render's RENDER_CREDITS. */\nconst RENDER_CREDITS = 5;\n/** The deadline \u2014 the SAME number as request-render's RENDER_DEADLINE_MS. */\nconst RENDER_DEADLINE_MS = 1_800_000;\n/** A row is the re-driver's only once request-render has had time to start it. */\nconst REDRIVE_AFTER_MS = 30_000;\n/** Rows started per tick (bounded: a tick is one function invocation). */\nconst MAX_REDRIVE_PER_TICK = 20;\n/** A row still waiting for a run after this long is failed (nothing is held). */\nconst BACKLOG_MAX_AGE_MS = 3_600_000;\n\ntype Env = CronFunctionEnvelope;\ntype RenderRow = {\n item_id: string;\n created_at?: string;\n data: {\n render_key?: string; request_key?: string; owner?: string; composition?: string;\n props?: Record<string, unknown>; created_at?: string; redrive_attempts?: number;\n };\n};\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n const notifications = env.scoped_jwts?.notifications;\n const jobsKey = env.secrets?.vxil_jobs_key;\n const renderUrl = env.secrets?.render_url;\n const renderToken = env.secrets?.render_token;\n if (!cms || !notifications) return Response.json({ error: 'missing cms/notifications scope' }, { status: 403 });\n if (!jobsKey || !renderUrl || !renderToken) {\n return Response.json(\n { error: 'store the secrets: vxil secrets set functions/vxil_jobs_key (an API key holding only jobs:write), functions/render_url, functions/render_token' },\n { status: 503 },\n );\n }\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n const now = Date.now();\n\n // the oldest pending rows with no run: `status` (s4) and `created_at` (t1)\n // are index slots, so the slots pick the rows and the `run_id: null` test\n // only runs over what they picked\n const filter = encodeURIComponent(JSON.stringify({\n status: 'pending',\n created_at: { $lt: new Date(now - REDRIVE_AFTER_MS).toISOString() },\n run_id: null,\n }));\n const listed = await fetch(\n `${base}/v1/cms/items/renders?filter=${filter}&sort=created_at&limit=${MAX_REDRIVE_PER_TICK}`,\n { headers: H },\n );\n if (!listed.ok) return Response.json({ error: `renders read: ${listed.status}` }, { status: 502 });\n const rows = ((await listed.json()) as { data?: { items?: RenderRow[] } }).data?.items ?? [];\n\n const out = { scanned: rows.length, started: 0, failed: 0, retry_later: 0, given_up: 0, unwritable: 0,\n notified: 0, notify_failed: 0,\n stopped: null as null | { status: number; code: string | null; retry_after: string | null } };\n const tell = async (row: RenderRow, why: 'credits' | 'waited') => {\n if (await notifyNotStarted(base, notifications, row, why)) out.notified += 1;\n else out.notify_failed += 1;\n };\n\n for (const row of rows) {\n const d = row.data;\n const attempts = (typeof d.redrive_attempts === 'number' ? d.redrive_attempts : 0) + 1;\n const createdAt = Date.parse(d.created_at ?? row.created_at ?? '');\n const mark = { redrive_attempts: attempts, redriven_at: new Date(now).toISOString() };\n\n // a row that cannot be a render request-render made (a hand-written row\n // missing its key parts), or one that has waited too long: fail it \u2014 no\n // run exists, so nothing is held and nothing is refunded\n const shapeOk = !!d.owner && !!d.request_key && !!d.composition\n && d.render_key === `${d.owner}:${d.request_key}`\n && d.composition.length <= 120 && JSON.stringify(d.props ?? {}).length <= 16_384;\n if (!shapeOk || (Number.isFinite(createdAt) && now - createdAt > BACKLOG_MAX_AGE_MS)) {\n const error = shapeOk\n ? 'not_started: the render waited too long for a free slot'\n : 'not_started: the row is not a render request-render created';\n // only while it is STILL pending with no run (a racing start wins)\n if (await patchRow(base, H, row.item_id, { ...mark, status: 'failed', error }, { status: 'pending', run_id: null })) {\n out.given_up += 1;\n // the owner last heard \"queued\" (a 429): say it will not happen. A\n // malformed row's owner is not trusted, so it is only failed.\n if (shapeOk) await tell(row, 'waited');\n } else if (!shapeOk) {\n out.unwritable += 1; // header note: fix it by hand\n }\n continue;\n }\n\n const enq = await startRun(base, jobsKey, renderUrl, renderToken, {\n itemId: row.item_id, user: d.owner!, renderKey: d.render_key!, requestKey: d.request_key!,\n composition: d.composition!, props: d.props ?? {},\n });\n if (!enq) { // network: try again next tick\n await patchRow(base, H, row.item_id, mark);\n out.retry_later += 1;\n continue;\n }\n if (enq.status === 402) {\n if (await patchRow(base, H, row.item_id, { ...mark, status: 'failed', error: 'insufficient_credits' })) await tell(row, 'credits');\n out.failed += 1;\n continue;\n }\n if (enq.status === 429 || (enq.status >= 400 && enq.status < 500)) {\n // 429: at the cap \u2014 the rest of the batch would get the same answer.\n // 400 / 401 / 403 / 422: a setup error (header note) \u2014 failing rows for\n // it would be wrong and could not be undone. Either way: record the\n // try, stop, and let the next tick (Retry-After: seconds) go on.\n await patchRow(base, H, row.item_id, mark);\n const code = enq.status === 429 ? null\n : ((await enq.json().catch(() => ({}))) as { error?: { code?: string } }).error?.code ?? null;\n out.stopped = { status: enq.status, code, retry_after: enq.headers.get('retry-after') };\n break;\n }\n if (!enq.ok) { // 5xx: try again next tick\n await patchRow(base, H, row.item_id, mark);\n out.retry_later += 1;\n continue;\n }\n const run = ((await enq.json()) as { data: { run_id: string; generation_status?: string; deduplicated?: boolean } }).data;\n if (run.deduplicated && run.generation_status === 'failed') {\n // the run already ENDED (its row write was lost): record it as failed\n await patchRow(base, H, row.item_id, { ...mark, run_id: run.run_id, status: 'failed' });\n out.failed += 1;\n continue;\n }\n await patchRow(base, H, row.item_id, { ...mark, run_id: run.run_id });\n out.started += 1;\n }\n // a 2xx either way: a cron tick is never retried (the next tick is the retry)\n return Response.json(out);\n },\n};\n\ninterface Start {\n itemId: string; user: string; renderKey: string; requestKey: string;\n composition: string; props: Record<string, unknown>;\n}\n\n/** The SAME webhook-mode generation request-render starts (the CI gate keeps\n * the two descriptors equal), sent with the jobs:write server key. Null on a\n * network error. */\nasync function startRun(base: string, key: string, renderUrl: string, renderToken: string, a: Start): Promise<Response | null> {\n return fetch(`${base}/v1/jobs/generation`, {\n // tenant-key: jobs:write via secret:vxil_jobs_key \u2014 called with the server\n // key, not the function's scoped token (header note); the CI gate checks\n // the secret is declared instead of a jobs scope\n method: 'POST',\n headers: { authorization: `Bearer ${key}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n job_name: 'render',\n provider: {\n url: renderUrl,\n method: 'POST',\n headers: { authorization: `Bearer ${renderToken}` },\n body: { render_id: a.itemId, user_id: a.user, composition: a.composition, props: a.props },\n },\n completion: { mode: 'webhook', status_path: 'status' },\n status_mirror: {\n feature: 'cms', collection: 'renders', record_id: a.itemId, column: 'status',\n progress_fields: ['progress', 'stage', 'message'],\n },\n reserve_credits: { amount: RENDER_CREDITS, user_id: a.user, credit_type: 'render_credits', reason: `render ${a.composition}` },\n timeout: { after_ms: RENDER_DEADLINE_MS },\n payload: {\n generation_id: a.itemId, correlation_id: a.requestKey,\n deadline_at: new Date(Date.now() + RENDER_DEADLINE_MS).toISOString(),\n },\n idempotency_key: a.renderKey,\n }),\n }).catch(() => null);\n}\n\n/** PATCH a render row (optionally only while `when` still holds); true when it landed. */\nasync function patchRow(\n base: string, H: Record<string, string>, itemId: string,\n data: Record<string, unknown>, when?: Record<string, unknown>,\n): Promise<boolean> {\n const res = await fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(itemId)}`, {\n method: 'PATCH',\n headers: H,\n body: JSON.stringify(when ? { data, if: when } : { data }),\n }).catch(() => undefined);\n return res?.ok ?? false;\n}\n\n/** Tell the owner a render they were told was queued will not start. One\n * message per render (Idempotency-Key render-not-started:<item_id> \u2014 the SAME\n * key notify-ready uses for a re-driven credit refusal, so the two never both\n * send). True when it was accepted. */\nasync function notifyNotStarted(base: string, token: string, row: RenderRow, why: 'credits' | 'waited'): Promise<boolean> {\n const name = row.data.composition ?? 'Your render';\n const res = await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: {\n authorization: `Bearer ${token}`, 'content-type': 'application/json',\n 'idempotency-key': `render-not-started:${row.item_id}`,\n },\n body: JSON.stringify({\n user_id: row.data.owner,\n template: 'transactional',\n data: why === 'credits'\n ? { subject: 'Your render could not start', paragraph: `\"${name}\" was queued, but there were not enough credits when its turn came. Nothing was charged \u2014 top up and try again.` }\n : { subject: 'Your render could not start', paragraph: `\"${name}\" waited over an hour for a free slot and was cancelled. Nothing was charged \u2014 try again later.` },\n }),\n }).catch(() => undefined);\n return res?.ok ?? false;\n}\n",
18450
- "request-render.ts": "// request-render.ts \u2014 start ONE long render for the signed-in user (a vxil\n// function, http trigger, END-USER mode).\n//\n// POST /v1/fn/request-render (with the user's session)\n// { \"composition\": \"promo-30s\", \"request_key\": \"<your idempotency key>\", \"props\": { \u2026 } }\n// \u2192 202 { item_id, run_id, credits } a new render, credits held\n// \u2192 200 { duplicate: true, request_key, item_id, run_id, status }\n// this user's request_key already started one\n// (or its run has already ended: status 'failed')\n// \u2192 402 { error: 'insufficient_credits', item_id } nothing held; use a new request_key\n// \u2192 429 / 502 { error, queued: true, item_id, request_key, retry_after }\n// no run YET: the row is queued and\n// redrive-pending starts it (credits\n// held then). Never retry with a NEW\n// request_key \u2014 that is a second render.\n//\n// What happens after the 202 is vxil's and your runtime's, not this function's:\n// \u2022 the generation lane POSTs your render endpoint (secret render_url) with\n// `Authorization: Bearer <render_token>` and the JSON body\n// { render_id, user_id, composition, props,\n// payload: { generation_id, correlation_id, deadline_at }, callback_url }\n// (plus an X-Vxil-Jobs-Signature header). The endpoint must answer 2xx\n// within 20 s \u2014 queue the work, do not render inline. A 408 / 429 / 5xx or\n// a timeout is retried; any other 4xx ends the run and refunds the credits.\n// \u2022 your runtime POSTs JSON to callback_url: { \"status\": \"processing\", \u2026 }\n// while it works, then { \"status\": \"completed\", \"output_url\": \"\u2026\",\n// \"duration_s\": 31.2, \"progress\": 100, \"stage\": \"done\" } \u2014 or\n// { \"status\": \"failed\", \"error\": \"render_failed\", \"hint\": \"\u2026\" }.\n// \u2022 vxil settles: completed COMMITS the held credits and writes every key of\n// the body onto this render's row; failed, no answer by the deadline, or a\n// cancel REFUNDS them and the row says failed. `deadline_at` (ISO time) is\n// when vxil stops waiting: a runtime that cannot finish by then should post\n// `failed` itself and stop \u2014 a `completed` after it changes nothing (the\n// credits are already back and the row says failed).\n// Every run ends with job.generation.completed | failed (generation_id = the\n// row's item_id, correlation_id = its request_key) \u2014 notify-ready listens.\n//\n// DELIVERY IS AT-LEAST-ONCE and users double-tap: the row's render_key\n// (owner + ':' + request_key \u2014 per user) is unique, so a second start with the\n// same key is a 409. On a 409 we read THIS user's row: a row that already has\n// its run is a duplicate; a row with no run (the first start died between the\n// row and the enqueue, or lost the enqueue's answer) is RE-DRIVEN \u2014 the run\n// carries the same render_key as its idempotency_key, so jobs hands back the\n// existing run instead of starting a second render.\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\ntype Input = { composition?: string; request_key?: string; props?: Record<string, unknown> };\ntype Env = HttpFunctionEnvelope<Input>;\ntype RenderRow = {\n item_id: string;\n data: { composition?: string; props?: Record<string, unknown>; run_id?: string; status?: string };\n};\n\n/** What one render costs, in `render_credits`. Price by composition if yours\n * differ \u2014 the per-run hold is clamped to `generation.maxReserveCredits`. */\nconst RENDER_CREDITS = 5;\n/** The deadline: a render your runtime never reports on fails and refunds\n * after this long. At most one hour (`generation.maxTimeoutMs`). */\nconst RENDER_DEADLINE_MS = 1_800_000;\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n const jobs = env.scoped_jwts?.jobs;\n if (!cms || !jobs) return Response.json({ error: 'missing cms/jobs scope' }, { status: 403 });\n const renderUrl = env.secrets?.render_url;\n const renderToken = env.secrets?.render_token;\n if (!renderUrl || !renderToken) {\n return Response.json(\n { error: 'store your render endpoint: vxil secrets set functions/render_url and functions/render_token' },\n { status: 500 },\n );\n }\n // the held credits are FORCED onto the verified end-user\n const user = env.end_user?.id;\n if (!user) return Response.json({ error: \"invoke request-render with the user's session (end-user mode)\" }, { status: 401 });\n\n const composition = typeof env.payload?.composition === 'string' ? env.payload.composition.trim().slice(0, 120) : '';\n const requestKey = typeof env.payload?.request_key === 'string' ? env.payload.request_key.slice(0, 120) : '';\n const rawProps = env.payload?.props;\n const props = rawProps && typeof rawProps === 'object' && !Array.isArray(rawProps) ? rawProps : {};\n if (!composition || !requestKey) {\n return Response.json({ error: 'need { composition, request_key, props? }' }, { status: 422 });\n }\n if (JSON.stringify(props).length > 16_384) {\n return Response.json({ error: 'props too large (16 KB max) \u2014 pass a reference to your own storage instead' }, { status: 413 });\n }\n const renderKey = `${user}:${requestKey}`;\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n\n // 1. create-or-find the row the app watches (owned by the user \u2014 cms forces\n // `owner` in end-user mode; the render_key_shape hook checks the key)\n const created = await fetch(`${base}/v1/cms/items/renders`, {\n method: 'POST',\n headers: H,\n body: JSON.stringify({\n data: {\n render_key: renderKey, request_key: requestKey, owner: user, composition, props,\n credits: RENDER_CREDITS, status: 'pending', created_at: new Date().toISOString(),\n },\n }),\n });\n if (created.status === 409) {\n // THIS user's row for this key (the read is owner-scoped in end-user mode)\n const filter = encodeURIComponent(JSON.stringify({ render_key: renderKey }));\n const found = await fetch(`${base}/v1/cms/items/renders?filter=${filter}&limit=1`, { headers: H });\n const row = found.ok ? ((await found.json()) as { data?: { items?: RenderRow[] } }).data?.items?.[0] : undefined;\n if (!row) return Response.json({ error: `render lookup: ${found.status}` }, { status: 502 });\n if (row.data.run_id || row.data.status === 'failed') {\n return Response.json({\n duplicate: true, request_key: requestKey, item_id: row.item_id,\n run_id: row.data.run_id ?? null, status: row.data.status ?? 'pending',\n });\n }\n // a start that never got its run: re-drive it (jobs dedupes on render_key)\n return startRun({\n base, H, jobs, renderUrl, renderToken, user, renderKey, requestKey,\n itemId: row.item_id, composition: row.data.composition ?? composition, props: row.data.props ?? props,\n });\n }\n if (!created.ok) return Response.json({ error: `render row: ${created.status}` }, { status: 502 });\n const itemId = ((await created.json()) as { data: { item_id: string } }).data.item_id;\n return startRun({ base, H, jobs, renderUrl, renderToken, user, renderKey, requestKey, itemId, composition, props });\n },\n};\n\ninterface StartArgs {\n base: string; H: Record<string, string>; jobs: string; renderUrl: string; renderToken: string;\n user: string; renderKey: string; requestKey: string; itemId: string;\n composition: string; props: Record<string, unknown>;\n}\n\n/** 2. the webhook-mode generation run: your endpoint, the held credits, the\n * deadline and the status mirror. Idempotent on render_key: a re-drive gets\n * the run that already exists. */\nasync function startRun(a: StartArgs): Promise<Response> {\n const enq = await fetch(`${a.base}/v1/jobs/generation`, {\n method: 'POST',\n headers: { authorization: `Bearer ${a.jobs}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n job_name: 'render',\n provider: {\n url: a.renderUrl,\n method: 'POST',\n // stored with the run for your project only, shown as [redacted] on\n // every run read, sent to your endpoint on the start call\n headers: { authorization: `Bearer ${a.renderToken}` },\n // your endpoint receives this, plus `payload` and `callback_url`\n // (user_id: the owner a coordinator uploads the output file for)\n body: { render_id: a.itemId, user_id: a.user, composition: a.composition, props: a.props },\n },\n completion: { mode: 'webhook', status_path: 'status' },\n // the status word onto `status`, and a `processing` ping's progress /\n // stage / message onto the same-named fields of the row\n status_mirror: {\n feature: 'cms', collection: 'renders', record_id: a.itemId, column: 'status',\n progress_fields: ['progress', 'stage', 'message'],\n },\n reserve_credits: { amount: RENDER_CREDITS, user_id: a.user, credit_type: 'render_credits', reason: `render ${a.composition}` },\n timeout: { after_ms: RENDER_DEADLINE_MS },\n // rides job.generation.* as generation_id / correlation_id, and reaches\n // your endpoint beside callback_url. deadline_at = when vxil stops\n // waiting (the deadline counts from this enqueue; a re-drive gets the\n // first run back, with ITS payload).\n payload: {\n generation_id: a.itemId, correlation_id: a.requestKey,\n deadline_at: new Date(Date.now() + RENDER_DEADLINE_MS).toISOString(),\n },\n idempotency_key: a.renderKey,\n }),\n });\n if (enq.status === 402) {\n // not enough credits: the run already ENDED (job.generation.failed,\n // ReserveInsufficient) and nothing was held. This key is spent; a new\n // attempt (after a top-up) uses a new request_key.\n const marked = await patchRow(a, { status: 'failed', error: 'insufficient_credits' });\n // if that write failed the row still says pending with no run; a retry with\n // the same key re-drives, gets the ended run back and marks it failed then\n return Response.json(\n { error: 'insufficient_credits', item_id: a.itemId, ...(marked ? {} : { row_updated: false }) },\n { status: 402 },\n );\n }\n if (!enq.ok) {\n // 429 (too many in flight) / 5xx: no run YET \u2014 the row stays pending with\n // no run, and it is QUEUED: redrive-pending (cron) starts it once a slot\n // frees up (within the hour, or the row is failed and the owner told) and\n // holds the credits then. `queued: true` says so. The app must NOT retry\n // with a NEW request_key (that is a second render, charged twice): show\n // \"queued\", watch the row, and retry only with the SAME request_key.\n return Response.json(\n { error: `generation enqueue: ${enq.status}`, queued: true, item_id: a.itemId, request_key: a.requestKey, retry_after: enq.headers.get('retry-after') },\n { status: enq.status === 429 ? 429 : 502 },\n );\n }\n const run = ((await enq.json()) as { data: { run_id: string; generation_status?: string; deduplicated?: boolean } }).data;\n if (run.deduplicated && run.generation_status === 'failed') {\n // a re-drive whose run already ENDED (refused for credits, failed or timed\n // out, and the row write that said so was lost): record it and say so \u2014\n // never report a fresh render with credits held\n const marked = await patchRow(a, { run_id: run.run_id, status: 'failed' });\n return Response.json({\n duplicate: true, request_key: a.requestKey, item_id: a.itemId, run_id: run.run_id, status: 'failed',\n ...(marked ? {} : { row_updated: false }),\n });\n }\n const linked = await patchRow(a, { run_id: run.run_id });\n // the run exists either way (and settles the row through the status mirror);\n // an unlinked row is linked by the next same-key call\n return Response.json(\n { item_id: a.itemId, run_id: run.run_id, credits: RENDER_CREDITS, ...(linked ? {} : { row_updated: false }) },\n { status: 202 },\n );\n}\n\n/** PATCH this render's row; true when the write landed. */\nasync function patchRow(a: StartArgs, data: Record<string, unknown>): Promise<boolean> {\n const res = await fetch(`${a.base}/v1/cms/items/renders/${encodeURIComponent(a.itemId)}`, {\n method: 'PATCH',\n headers: a.H,\n body: JSON.stringify({ data }),\n }).catch(() => undefined);\n return res?.ok ?? false;\n}\n"
18567
+ "notify-ready.ts": "// notify-ready.ts \u2014 job.generation.completed | failed \u2192 tell the render's owner\n// (a vxil function, webhook trigger on `job.generation.`).\n//\n// The status mirror has already written the row (status, and on completion\n// every key your runtime sent back). This function adds the two things a\n// mirror cannot: the failure cause on the row, and a message to the user.\n//\n// AT-LEAST-ONCE: an event can be delivered again. And because the trigger\n// declares `retry: { maxAttempts: 3 }`, a non-2xx answer from here goes back\n// to the jobs ladder for another attempt (without `retry` it would simply be\n// acknowledged). Every attempt carries the same event. The notification carries an\n// Idempotency-Key of one per RUN, so a redelivery sends nothing twice, and the\n// row patch writes the same value again.\n\nimport type { JobGenerationSettledEventPayload, WebhookFunctionEnvelope } from '@vxil/sdk';\n\ntype RenderRow = {\n owner?: string; composition?: string; output_url?: string; error?: string; redrive_attempts?: number;\n run_id?: string | null; status?: string;\n};\n\n/** Platform error classes settle with no hint; these are the ones a render\n * can end with, in words for the row (the class stays first, for code). */\nconst PLATFORM_HINTS: Record<string, string> = {\n GenerationExpired: 'the render did not finish before its deadline',\n RetryableHttp: 'the render endpoint kept failing to accept the render (retries exhausted)',\n NetworkError: 'the render endpoint could not be reached (retries exhausted)',\n};\n\nfunction patchError(base: string, H: Record<string, string>, id: string, error: string): Promise<Response> {\n return fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(id)}`, {\n method: 'PATCH', headers: H, body: JSON.stringify({ data: { error } }),\n });\n}\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as WebhookFunctionEnvelope<JobGenerationSettledEventPayload>;\n const d = env.payload?.data;\n // job.generation.queued carries no status; a truncated event has no fields\n if (!d || 'truncated' in d || !('status' in d) || !d.generation_id) return Response.json({ skipped: true });\n const cms = env.scoped_jwts?.cms;\n const notifications = env.scoped_jwts?.notifications;\n if (!cms || !notifications) return Response.json({ error: 'missing cms/notifications scope' }, { status: 403 });\n const base = env.vxil_base;\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n\n const rowRes = await fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(d.generation_id)}`, { headers: H });\n // another job's generation event (not a render) \u2014 nothing to do\n if (rowRes.status === 404) return Response.json({ skipped: 'not a render' });\n if (!rowRes.ok) return Response.json({ error: `render read: ${rowRes.status}` }, { status: 502 }); // another attempt (header note)\n const row = ((await rowRes.json()) as { data: { data: RenderRow } }).data.data;\n if (!row.owner) return Response.json({ skipped: 'no owner' });\n\n if (d.status === 'failed') {\n // BRANCH ON error_class \u2014 not every `failed` is a render that failed:\n // \u2022 PaymentsUnavailable: the credits could not be held when the run was\n // started, so it ended before it began (nothing held). The start\n // answered 503 and the row is still pending \u2014 redrive-pending starts\n // it on a later tick (a NEW run). Nothing to tell the owner.\n // ONE race to undo: a request-render that re-sent the same key while\n // that start was still waiting on payments was handed THIS run (202,\n // deduplicated) and linked it \u2014 and redrive-pending only picks rows\n // with no run_id. Unlink it (only while the row is still pending on\n // this very run), so the re-driver starts it.\n if (d.error_class === 'PaymentsUnavailable') {\n if (row.run_id === d.run_id && (row.status ?? 'pending') === 'pending') {\n const unlinked = await fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(d.generation_id)}`, {\n method: 'PATCH', headers: H,\n body: JSON.stringify({ data: { run_id: null }, if: { run_id: d.run_id, status: 'pending' } }),\n });\n // 409: the row moved on meanwhile (re-driven, failed) \u2014 nothing to undo\n if (!unlinked.ok && unlinked.status !== 409) {\n return Response.json({ error: `render patch: ${unlinked.status}` }, { status: 502 }); // another attempt (header note)\n }\n return Response.json({ skipped: 'not started: payments unavailable', unlinked: unlinked.ok });\n }\n return Response.json({ skipped: 'not started: payments unavailable' });\n }\n // \u2022 Cancelled: your own cancel (POST /v1/jobs/runs/{id}/cancel); the\n // credits were released. The row says so; the owner is not told \"try\n // again\" \u2014 tell them from where you cancelled, if at all.\n if (d.error_class === 'Cancelled') {\n if (row.error !== 'cancelled') {\n const patched = await patchError(base, H, d.generation_id, 'cancelled');\n if (!patched.ok) return Response.json({ error: `render patch: ${patched.status}` }, { status: 502 }); // another attempt (header note)\n }\n return Response.json({ ok: true, notified: false });\n }\n // Not enough credits. A render request-render started itself already\n // answered the user (402) and wrote error: 'insufficient_credits' \u2014 keep\n // that code on the row (write it only if that write was lost) and send\n // nothing. A render the RE-DRIVER started is different: the user last\n // heard 429 \"queued\", so they are told \u2014 with the re-driver's own key\n // (render-not-started:<item_id>), so the two never both send.\n if (d.error_class === 'ReserveInsufficient') {\n if (!row.error) {\n const patched = await patchError(base, H, d.generation_id, 'insufficient_credits');\n if (!patched.ok) return Response.json({ error: `render patch: ${patched.status}` }, { status: 502 }); // another attempt (header note)\n }\n if (!(typeof row.redrive_attempts === 'number' && row.redrive_attempts > 0)) {\n return Response.json({ ok: true, notified: false });\n }\n return send(base, notifications, `render-not-started:${d.generation_id}`, row.owner, {\n subject: 'Your render could not start',\n paragraph: `\"${row.composition ?? 'Your render'}\" was queued, but there were not enough credits when its turn came. Nothing was charged \u2014 top up and try again.`,\n });\n }\n // the cause the run settled with: your runtime's `error` (+ `hint`), or\n // the platform's own class \u2014 which carries no hint, so a short human\n // one is added for the ones a user can meet (PLATFORM_HINTS)\n const hint = d.error_hint ?? (d.error_class ? PLATFORM_HINTS[d.error_class] : undefined);\n const cause = [d.error_class, hint].filter(Boolean).join(': ') || 'render failed';\n if (row.error !== cause) {\n const patched = await patchError(base, H, d.generation_id, cause.slice(0, 500));\n if (!patched.ok) return Response.json({ error: `render patch: ${patched.status}` }, { status: 502 }); // another attempt (header note)\n }\n }\n\n return send(base, notifications, `render-ready:${d.run_id}`, row.owner, d.status === 'completed'\n ? { subject: 'Your render is ready', paragraph: `\"${row.composition ?? 'Your render'}\" finished. Open the app to watch it.` }\n : { subject: 'Your render could not finish', paragraph: 'Nothing was charged \u2014 the credits are back on your balance. Try again in a minute.' });\n },\n};\n\n/** One transactional message; a non-2xx answer asks the ladder for another attempt. */\nasync function send(\n base: string, token: string, idempotencyKey: string, userId: string,\n data: { subject: string; paragraph: string },\n): Promise<Response> {\n const sent = await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json', 'idempotency-key': idempotencyKey },\n body: JSON.stringify({ user_id: userId, template: 'transactional', data }),\n });\n return sent.ok\n ? Response.json({ ok: true, notified: true })\n : Response.json({ error: `notifications send: ${sent.status}` }, { status: 502 }); // another attempt (header note)\n}\n",
18568
+ "redrive-pending.ts": "// redrive-pending.ts \u2014 the BACKLOG RE-DRIVER (a vxil function, cron trigger,\n// every minute; `overlap: 'skip'`).\n//\n// cron-walk: drains-filter \u2014 every row it starts is PATCHed with its run_id (and\n// every row it gives up on to status 'failed'), which takes it out of the\n// `{ status: 'pending', run_id: null }` read; a 429 stops the tick early.\n//\n// Why it exists: at the generation concurrency cap (`generation.maxConcurrent`\n// open runs \u2014 20 in this blueprint, up to 200) request-render answers 429 and\n// leaves the row `pending` with NO run. Without this function only the caller\n// re-drives it (the same request_key again). With it, a backlog drains on its\n// own: each tick reads the oldest pending rows that have had no run for at\n// least REDRIVE_AFTER_MS, and starts each one with the SAME descriptor and the\n// SAME idempotency key (the row's render_key) request-render uses \u2014 so a row a\n// user is re-driving at the same moment still gets exactly one run.\n//\n// per row, by the jobs answer:\n// 202 (new run, or an open one handed back) \u2192 PATCH run_id (the status\n// mirror settles the row from here)\n// 202 deduplicated, generation_status 'failed' \u2192 PATCH run_id + status 'failed'\n// 402 (too few credits) \u2192 PATCH status 'failed',\n// error 'insufficient_credits',\n// and TELL the owner (they last\n// heard \"queued\", not \"refused\")\n// 429 (cap reached / too many credits held), \u2192 STOP the tick; the rest wait\n// or 503 payments_unavailable (credits could for the next one (the wait\n// not be held: nothing started or held) header is reported; the same\n// key starts a fresh run then)\n// 400 / 401 / 403 / 422 \u2192 STOP the tick, report it. Every\n// re-driven row sends the same\n// descriptor apart from its own\n// (already bounded) composition and\n// props, so a refusal is a SETUP\n// error \u2014 a non-https or private\n// render_url, a wrong or revoked\n// key, a config change \u2014 that\n// would fail every row the same\n// way. Nothing is failed for it:\n// fix the setup and the next tick\n// carries on.\n// 5xx / network \u2192 record the attempt, go on\n// A row still pending with no run after BACKLOG_MAX_AGE_MS is the ONLY thing\n// the re-driver gives up on: status 'failed', error 'not_started: \u2026' (nothing\n// was ever held for it), and the owner is told once\n// (Idempotency-Key render-not-started:<item_id>).\n// Every try is recorded on the row: redrive_attempts, redriven_at.\n//\n// A HAND-WRITTEN ROW that breaks render_key_shape (written before the hook\n// existed, or by a path that bypassed it) cannot be patched either \u2014 the\n// hook judges the merged row \u2014 so it would stay pending and take one of the\n// tick's slots for good. Fix or delete such rows by hand; the tick reports\n// them as `unwritable`.\n\n// THE KEY: a function-originated credit hold must name the user it acts for,\n// and a cron tick has no signed-in user \u2014 so the jobs call is made with\n// `vxil_jobs_key`, an API key of this same backend holding ONLY `jobs:write`\n// (a trusted server key may hold credits for any of your users; README \"The\n// backlog\"). The rows are read and written with the function's own scoped cms\n// token. The credits held are this blueprint's fixed price, never the row's\n// `credits` field \u2014 a signed-in user can edit their own row, so nothing the\n// re-driver charges or starts is read from a value they could have lowered.\n// The owner is told with the function's scoped notifications token.\n\nimport type { CronFunctionEnvelope } from '@vxil/sdk';\n\n/** What one render costs \u2014 the SAME number as request-render's RENDER_CREDITS. */\nconst RENDER_CREDITS = 5;\n/** The deadline \u2014 the SAME number as request-render's RENDER_DEADLINE_MS. */\nconst RENDER_DEADLINE_MS = 1_800_000;\n/** A row is the re-driver's only once request-render has had time to start it. */\nconst REDRIVE_AFTER_MS = 30_000;\n/** Rows started per tick (bounded: a tick is one function invocation). */\nconst MAX_REDRIVE_PER_TICK = 20;\n/** A row still waiting for a run after this long is failed (nothing is held). */\nconst BACKLOG_MAX_AGE_MS = 3_600_000;\n\ntype Env = CronFunctionEnvelope;\ntype RenderRow = {\n item_id: string;\n created_at?: string;\n data: {\n render_key?: string; request_key?: string; owner?: string; composition?: string;\n props?: Record<string, unknown>; created_at?: string; redrive_attempts?: number;\n };\n};\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n const notifications = env.scoped_jwts?.notifications;\n const jobsKey = env.secrets?.vxil_jobs_key;\n const renderUrl = env.secrets?.render_url;\n const renderToken = env.secrets?.render_token;\n if (!cms || !notifications) return Response.json({ error: 'missing cms/notifications scope' }, { status: 403 });\n if (!jobsKey || !renderUrl || !renderToken) {\n return Response.json(\n { error: 'store the secrets: vxil secrets set functions/vxil_jobs_key (an API key holding only jobs:write), functions/render_url, functions/render_token' },\n { status: 503 },\n );\n }\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n const now = Date.now();\n\n // the oldest pending rows with no run: `status` (s4) and `created_at` (t1)\n // are index slots, so the slots pick the rows and the `run_id: null` test\n // only runs over what they picked\n const filter = encodeURIComponent(JSON.stringify({\n status: 'pending',\n created_at: { $lt: new Date(now - REDRIVE_AFTER_MS).toISOString() },\n run_id: null,\n }));\n const listed = await fetch(\n `${base}/v1/cms/items/renders?filter=${filter}&sort=created_at&limit=${MAX_REDRIVE_PER_TICK}`,\n { headers: H },\n );\n if (!listed.ok) return Response.json({ error: `renders read: ${listed.status}` }, { status: 502 });\n const rows = ((await listed.json()) as { data?: { items?: RenderRow[] } }).data?.items ?? [];\n\n const out = { scanned: rows.length, started: 0, failed: 0, retry_later: 0, given_up: 0, unwritable: 0,\n notified: 0, notify_failed: 0,\n stopped: null as null | { status: number; code: string | null; retry_after: string | null } };\n const tell = async (row: RenderRow, why: 'credits' | 'waited') => {\n if (await notifyNotStarted(base, notifications, row, why)) out.notified += 1;\n else out.notify_failed += 1;\n };\n\n for (const row of rows) {\n const d = row.data;\n const attempts = (typeof d.redrive_attempts === 'number' ? d.redrive_attempts : 0) + 1;\n const createdAt = Date.parse(d.created_at ?? row.created_at ?? '');\n const mark = { redrive_attempts: attempts, redriven_at: new Date(now).toISOString() };\n\n // a row that cannot be a render request-render made (a hand-written row\n // missing its key parts), or one that has waited too long: fail it \u2014 no\n // run exists, so nothing is held and nothing is refunded\n const shapeOk = !!d.owner && !!d.request_key && !!d.composition\n && d.render_key === `${d.owner}:${d.request_key}`\n && d.composition.length <= 120 && JSON.stringify(d.props ?? {}).length <= 16_384;\n if (!shapeOk || (Number.isFinite(createdAt) && now - createdAt > BACKLOG_MAX_AGE_MS)) {\n const error = shapeOk\n ? 'not_started: the render waited too long for a free slot'\n : 'not_started: the row is not a render request-render created';\n // only while it is STILL pending with no run (a racing start wins)\n if (await patchRow(base, H, row.item_id, { ...mark, status: 'failed', error }, { status: 'pending', run_id: null })) {\n out.given_up += 1;\n // the owner last heard \"queued\" (a 429): say it will not happen. A\n // malformed row's owner is not trusted, so it is only failed.\n if (shapeOk) await tell(row, 'waited');\n } else if (!shapeOk) {\n out.unwritable += 1; // header note: fix it by hand\n }\n continue;\n }\n\n const enq = await startRun(base, jobsKey, renderUrl, renderToken, {\n itemId: row.item_id, user: d.owner!, renderKey: d.render_key!, requestKey: d.request_key!,\n composition: d.composition!, props: d.props ?? {},\n });\n if (!enq) { // network: try again next tick\n await patchRow(base, H, row.item_id, mark);\n out.retry_later += 1;\n continue;\n }\n if (enq.status === 402) {\n if (await patchRow(base, H, row.item_id, { ...mark, status: 'failed', error: 'insufficient_credits' })) await tell(row, 'credits');\n out.failed += 1;\n continue;\n }\n const code = enq.status === 429 || (enq.status >= 400 && enq.status !== 402)\n ? ((await enq.json().catch(() => ({}))) as { error?: { code?: string } }).error?.code ?? null\n : null;\n if (enq.status === 429 || (enq.status >= 400 && enq.status < 500) || code === 'payments_unavailable') {\n // 429: at the cap \u2014 the rest of the batch would get the same answer.\n // 503 payments_unavailable: the credits could not be held right now\n // (nothing was started or held) \u2014 the rest would get the same answer;\n // the next tick re-sends the SAME key and gets a fresh run.\n // 400 / 401 / 403 / 422: a setup error (header note) \u2014 failing rows for\n // it would be wrong and could not be undone. Either way: record the\n // try, stop, and let the next tick (Retry-After: seconds) go on.\n await patchRow(base, H, row.item_id, mark);\n out.stopped = { status: enq.status, code: enq.status === 429 ? null : code, retry_after: enq.headers.get('retry-after') };\n break;\n }\n if (!enq.ok) { // 5xx: try again next tick\n await patchRow(base, H, row.item_id, mark);\n out.retry_later += 1;\n continue;\n }\n const run = ((await enq.json()) as { data: { run_id: string; generation_status?: string; deduplicated?: boolean } }).data;\n if (run.deduplicated && run.generation_status === 'failed') {\n // the run already ENDED (its row write was lost): record it as failed\n await patchRow(base, H, row.item_id, { ...mark, run_id: run.run_id, status: 'failed' });\n out.failed += 1;\n continue;\n }\n await patchRow(base, H, row.item_id, { ...mark, run_id: run.run_id });\n out.started += 1;\n }\n // a 2xx either way: a cron tick is never retried (the next tick is the retry)\n return Response.json(out);\n },\n};\n\ninterface Start {\n itemId: string; user: string; renderKey: string; requestKey: string;\n composition: string; props: Record<string, unknown>;\n}\n\n/** The SAME webhook-mode generation request-render starts (the CI gate keeps\n * the two descriptors equal), sent with the jobs:write server key. Null on a\n * network error. */\nasync function startRun(base: string, key: string, renderUrl: string, renderToken: string, a: Start): Promise<Response | null> {\n return fetch(`${base}/v1/jobs/generation`, {\n // tenant-key: jobs:write via secret:vxil_jobs_key \u2014 called with the server\n // key, not the function's scoped token (header note); the CI gate checks\n // the secret is declared instead of a jobs scope\n method: 'POST',\n headers: { authorization: `Bearer ${key}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n job_name: 'render',\n provider: {\n url: renderUrl,\n method: 'POST',\n headers: { authorization: `Bearer ${renderToken}` },\n body: { render_id: a.itemId, user_id: a.user, composition: a.composition, props: a.props },\n },\n completion: { mode: 'webhook', status_path: 'status' },\n status_mirror: {\n feature: 'cms', collection: 'renders', record_id: a.itemId, column: 'status',\n progress_fields: ['progress', 'stage', 'message'],\n },\n reserve_credits: { amount: RENDER_CREDITS, user_id: a.user, credit_type: 'render_credits', reason: `render ${a.composition}` },\n timeout: { after_ms: RENDER_DEADLINE_MS },\n payload: {\n generation_id: a.itemId, correlation_id: a.requestKey,\n deadline_at: new Date(Date.now() + RENDER_DEADLINE_MS).toISOString(),\n },\n idempotency_key: a.renderKey,\n }),\n }).catch(() => null);\n}\n\n/** PATCH a render row (optionally only while `when` still holds); true when it landed. */\nasync function patchRow(\n base: string, H: Record<string, string>, itemId: string,\n data: Record<string, unknown>, when?: Record<string, unknown>,\n): Promise<boolean> {\n const res = await fetch(`${base}/v1/cms/items/renders/${encodeURIComponent(itemId)}`, {\n method: 'PATCH',\n headers: H,\n body: JSON.stringify(when ? { data, if: when } : { data }),\n }).catch(() => undefined);\n return res?.ok ?? false;\n}\n\n/** Tell the owner a render they were told was queued will not start. One\n * message per render (Idempotency-Key render-not-started:<item_id> \u2014 the SAME\n * key notify-ready uses for a re-driven credit refusal, so the two never both\n * send). True when it was accepted. */\nasync function notifyNotStarted(base: string, token: string, row: RenderRow, why: 'credits' | 'waited'): Promise<boolean> {\n const name = row.data.composition ?? 'Your render';\n const res = await fetch(`${base}/v1/notifications/send`, {\n method: 'POST',\n headers: {\n authorization: `Bearer ${token}`, 'content-type': 'application/json',\n 'idempotency-key': `render-not-started:${row.item_id}`,\n },\n body: JSON.stringify({\n user_id: row.data.owner,\n template: 'transactional',\n data: why === 'credits'\n ? { subject: 'Your render could not start', paragraph: `\"${name}\" was queued, but there were not enough credits when its turn came. Nothing was charged \u2014 top up and try again.` }\n : { subject: 'Your render could not start', paragraph: `\"${name}\" waited over an hour for a free slot and was cancelled. Nothing was charged \u2014 try again later.` },\n }),\n }).catch(() => undefined);\n return res?.ok ?? false;\n}\n",
18569
+ "request-render.ts": "// request-render.ts \u2014 start ONE long render for the signed-in user (a vxil\n// function, http trigger, END-USER mode).\n//\n// POST /v1/fn/request-render (with the user's session)\n// { \"composition\": \"promo-30s\", \"request_key\": \"<your idempotency key>\", \"props\": { \u2026 } }\n// \u2192 202 { item_id, run_id, credits } a new render, credits held\n// \u2192 200 { duplicate: true, request_key, item_id, run_id, status }\n// this user's request_key already started one\n// (or its run has already ended: status 'failed')\n// \u2192 402 { error: 'insufficient_credits', item_id } nothing held; use a new request_key\n// \u2192 429 / 503 / 502 { error, queued: true, item_id, request_key, retry_after }\n// no run YET (429 at the cap, 503 credits\n// could not be held right now \u2014 nothing\n// held): the row is queued and\n// redrive-pending starts it (credits\n// held then). Never retry with a NEW\n// request_key \u2014 that is a second render.\n//\n// What happens after the 202 is vxil's and your runtime's, not this function's:\n// \u2022 the generation lane POSTs your render endpoint (secret render_url) with\n// `Authorization: Bearer <render_token>` and the JSON body\n// { render_id, user_id, composition, props,\n// payload: { generation_id, correlation_id, deadline_at }, callback_url }\n// (plus an X-Vxil-Jobs-Signature header). The endpoint must answer 2xx\n// within 20 s \u2014 queue the work, do not render inline. A 408 / 429 / 5xx or\n// a timeout is retried; any other 4xx ends the run and refunds the credits.\n// \u2022 your runtime POSTs JSON to callback_url: { \"status\": \"processing\", \u2026 }\n// while it works, then { \"status\": \"completed\", \"output_url\": \"\u2026\",\n// \"duration_s\": 31.2, \"progress\": 100, \"stage\": \"done\" } \u2014 or\n// { \"status\": \"failed\", \"error\": \"render_failed\", \"hint\": \"\u2026\" }.\n// \u2022 vxil settles: completed COMMITS the held credits and writes every key of\n// the body onto this render's row; failed, no answer by the deadline, or a\n// cancel REFUNDS them and the row says failed. `deadline_at` (ISO time) is\n// when vxil stops waiting: a runtime that cannot finish by then should post\n// `failed` itself and stop \u2014 a `completed` after it changes nothing (the\n// credits are already back and the row says failed).\n// Every run ends with job.generation.completed | failed (generation_id = the\n// row's item_id, correlation_id = its request_key) \u2014 notify-ready listens.\n//\n// DELIVERY IS AT-LEAST-ONCE and users double-tap: the row's render_key\n// (owner + ':' + request_key \u2014 per user) is unique, so a second start with the\n// same key is a 409. On a 409 we read THIS user's row: a row that already has\n// its run is a duplicate; a row with no run (the first start died between the\n// row and the enqueue, or lost the enqueue's answer) is RE-DRIVEN \u2014 the run\n// carries the same render_key as its idempotency_key, so jobs hands back the\n// existing run instead of starting a second render.\n\nimport type { HttpFunctionEnvelope } from '@vxil/sdk';\n\ntype Input = { composition?: string; request_key?: string; props?: Record<string, unknown> };\ntype Env = HttpFunctionEnvelope<Input>;\ntype RenderRow = {\n item_id: string;\n data: { composition?: string; props?: Record<string, unknown>; run_id?: string; status?: string };\n};\n\n/** What one render costs, in `render_credits`. Price by composition if yours\n * differ \u2014 the per-run hold is clamped to `generation.maxReserveCredits`. */\nconst RENDER_CREDITS = 5;\n/** The deadline: a render your runtime never reports on fails and refunds\n * after this long. At most one hour (`generation.maxTimeoutMs`). */\nconst RENDER_DEADLINE_MS = 1_800_000;\n\nexport default {\n async fetch(req: Request): Promise<Response> {\n const env = (await req.json().catch(() => ({}))) as Env;\n const base = env.vxil_base ?? 'https://api.vxil.com';\n const cms = env.scoped_jwts?.cms;\n const jobs = env.scoped_jwts?.jobs;\n if (!cms || !jobs) return Response.json({ error: 'missing cms/jobs scope' }, { status: 403 });\n const renderUrl = env.secrets?.render_url;\n const renderToken = env.secrets?.render_token;\n if (!renderUrl || !renderToken) {\n return Response.json(\n { error: 'store your render endpoint: vxil secrets set functions/render_url and functions/render_token' },\n { status: 500 },\n );\n }\n // the held credits are FORCED onto the verified end-user\n const user = env.end_user?.id;\n if (!user) return Response.json({ error: \"invoke request-render with the user's session (end-user mode)\" }, { status: 401 });\n\n const composition = typeof env.payload?.composition === 'string' ? env.payload.composition.trim().slice(0, 120) : '';\n const requestKey = typeof env.payload?.request_key === 'string' ? env.payload.request_key.slice(0, 120) : '';\n const rawProps = env.payload?.props;\n const props = rawProps && typeof rawProps === 'object' && !Array.isArray(rawProps) ? rawProps : {};\n if (!composition || !requestKey) {\n return Response.json({ error: 'need { composition, request_key, props? }' }, { status: 422 });\n }\n if (JSON.stringify(props).length > 16_384) {\n return Response.json({ error: 'props too large (16 KB max) \u2014 pass a reference to your own storage instead' }, { status: 413 });\n }\n const renderKey = `${user}:${requestKey}`;\n const H = { authorization: `Bearer ${cms}`, 'content-type': 'application/json' };\n\n // 1. create-or-find the row the app watches (owned by the user \u2014 cms forces\n // `owner` in end-user mode; the render_key_shape hook checks the key)\n const created = await fetch(`${base}/v1/cms/items/renders`, {\n method: 'POST',\n headers: H,\n body: JSON.stringify({\n data: {\n render_key: renderKey, request_key: requestKey, owner: user, composition, props,\n credits: RENDER_CREDITS, status: 'pending', created_at: new Date().toISOString(),\n },\n }),\n });\n if (created.status === 409) {\n // THIS user's row for this key (the read is owner-scoped in end-user mode)\n const filter = encodeURIComponent(JSON.stringify({ render_key: renderKey }));\n const found = await fetch(`${base}/v1/cms/items/renders?filter=${filter}&limit=1`, { headers: H });\n const row = found.ok ? ((await found.json()) as { data?: { items?: RenderRow[] } }).data?.items?.[0] : undefined;\n if (!row) return Response.json({ error: `render lookup: ${found.status}` }, { status: 502 });\n if (row.data.run_id || row.data.status === 'failed') {\n return Response.json({\n duplicate: true, request_key: requestKey, item_id: row.item_id,\n run_id: row.data.run_id ?? null, status: row.data.status ?? 'pending',\n });\n }\n // a start that never got its run: re-drive it (jobs dedupes on render_key)\n return startRun({\n base, H, jobs, renderUrl, renderToken, user, renderKey, requestKey,\n itemId: row.item_id, composition: row.data.composition ?? composition, props: row.data.props ?? props,\n });\n }\n if (!created.ok) return Response.json({ error: `render row: ${created.status}` }, { status: 502 });\n const itemId = ((await created.json()) as { data: { item_id: string } }).data.item_id;\n return startRun({ base, H, jobs, renderUrl, renderToken, user, renderKey, requestKey, itemId, composition, props });\n },\n};\n\ninterface StartArgs {\n base: string; H: Record<string, string>; jobs: string; renderUrl: string; renderToken: string;\n user: string; renderKey: string; requestKey: string; itemId: string;\n composition: string; props: Record<string, unknown>;\n}\n\n/** 2. the webhook-mode generation run: your endpoint, the held credits, the\n * deadline and the status mirror. Idempotent on render_key: a re-drive gets\n * the run that already exists. */\nasync function startRun(a: StartArgs): Promise<Response> {\n const enq = await fetch(`${a.base}/v1/jobs/generation`, {\n method: 'POST',\n headers: { authorization: `Bearer ${a.jobs}`, 'content-type': 'application/json' },\n body: JSON.stringify({\n job_name: 'render',\n provider: {\n url: a.renderUrl,\n method: 'POST',\n // stored with the run for your project only, shown as [redacted] on\n // every run read, sent to your endpoint on the start call\n headers: { authorization: `Bearer ${a.renderToken}` },\n // your endpoint receives this, plus `payload` and `callback_url`\n // (user_id: the owner a coordinator uploads the output file for)\n body: { render_id: a.itemId, user_id: a.user, composition: a.composition, props: a.props },\n },\n completion: { mode: 'webhook', status_path: 'status' },\n // the status word onto `status`, and a `processing` ping's progress /\n // stage / message onto the same-named fields of the row\n status_mirror: {\n feature: 'cms', collection: 'renders', record_id: a.itemId, column: 'status',\n progress_fields: ['progress', 'stage', 'message'],\n },\n reserve_credits: { amount: RENDER_CREDITS, user_id: a.user, credit_type: 'render_credits', reason: `render ${a.composition}` },\n timeout: { after_ms: RENDER_DEADLINE_MS },\n // rides job.generation.* as generation_id / correlation_id, and reaches\n // your endpoint beside callback_url. deadline_at = when vxil stops\n // waiting (the deadline counts from this enqueue; a re-drive gets the\n // first run back, with ITS payload).\n payload: {\n generation_id: a.itemId, correlation_id: a.requestKey,\n deadline_at: new Date(Date.now() + RENDER_DEADLINE_MS).toISOString(),\n },\n idempotency_key: a.renderKey,\n }),\n });\n if (enq.status === 402) {\n // not enough credits: the run already ENDED (job.generation.failed,\n // ReserveInsufficient) and nothing was held. This key is spent; a new\n // attempt (after a top-up) uses a new request_key.\n const marked = await patchRow(a, { status: 'failed', error: 'insufficient_credits' });\n // if that write failed the row still says pending with no run; a retry with\n // the same key re-drives, gets the ended run back and marks it failed then\n return Response.json(\n { error: 'insufficient_credits', item_id: a.itemId, ...(marked ? {} : { row_updated: false }) },\n { status: 402 },\n );\n }\n if (!enq.ok) {\n // 429 (too many in flight) / 503 payments_unavailable (the credits could\n // not be held right now \u2014 nothing was started or held; the same key starts\n // a fresh run later) / 5xx: no run YET \u2014 the row stays pending with\n // no run, and it is QUEUED: redrive-pending (cron) starts it once a slot\n // frees up (within the hour, or the row is failed and the owner told) and\n // holds the credits then. `queued: true` says so. The app must NOT retry\n // with a NEW request_key (that is a second render, charged twice): show\n // \"queued\", watch the row, and retry only with the SAME request_key.\n return Response.json(\n { error: `generation enqueue: ${enq.status}`, queued: true, item_id: a.itemId, request_key: a.requestKey, retry_after: enq.headers.get('retry-after') },\n { status: enq.status === 429 || enq.status === 503 ? enq.status : 502 },\n );\n }\n const run = ((await enq.json()) as { data: { run_id: string; generation_status?: string; deduplicated?: boolean } }).data;\n if (run.deduplicated && run.generation_status === 'failed') {\n // a re-drive whose run already ENDED (refused for credits, failed or timed\n // out, and the row write that said so was lost): record it and say so \u2014\n // never report a fresh render with credits held\n const marked = await patchRow(a, { run_id: run.run_id, status: 'failed' });\n return Response.json({\n duplicate: true, request_key: a.requestKey, item_id: a.itemId, run_id: run.run_id, status: 'failed',\n ...(marked ? {} : { row_updated: false }),\n });\n }\n const linked = await patchRow(a, { run_id: run.run_id });\n // the run exists either way (and settles the row through the status mirror);\n // an unlinked row is linked by the next same-key call\n return Response.json(\n { item_id: a.itemId, run_id: run.run_id, credits: RENDER_CREDITS, ...(linked ? {} : { row_updated: false }) },\n { status: 202 },\n );\n}\n\n/** PATCH this render's row; true when the write landed. */\nasync function patchRow(a: StartArgs, data: Record<string, unknown>): Promise<boolean> {\n const res = await fetch(`${a.base}/v1/cms/items/renders/${encodeURIComponent(a.itemId)}`, {\n method: 'PATCH',\n headers: a.H,\n body: JSON.stringify({ data }),\n }).catch(() => undefined);\n return res?.ok ?? false;\n}\n"
18451
18570
  }
18452
18571
  },
18453
18572
  {