@empire-builder-kit/nx 1.0.0-rc.39 → 1.0.0-rc.40

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@empire-builder-kit/nx",
3
- "version": "1.0.0-rc.39",
3
+ "version": "1.0.0-rc.40",
4
4
  "description": "Nx generators, executors, and Blueprint contracts for Empire Builder Kit",
5
5
  "keywords": [
6
6
  "aws",
@@ -145,7 +145,7 @@
145
145
  "provenance": false
146
146
  },
147
147
  "dependencies": {
148
- "@empire-builder-kit/runtime": "1.0.0-rc.39",
148
+ "@empire-builder-kit/runtime": "1.0.0-rc.40",
149
149
  "@aws-sdk/client-cloudfront": "^3.700.0",
150
150
  "@aws-sdk/client-cloudwatch": "^3.700.0",
151
151
  "@aws-sdk/client-cloudwatch-logs": "^3.700.0",
@@ -178,5 +178,5 @@
178
178
  "engines": {
179
179
  "node": ">=22.13.0"
180
180
  },
181
- "gitHead": "2875c7b517e99e6eab20c6efa31dfeb932b410f8"
181
+ "gitHead": "56a2e45b04b8ce8849990baba025cecb5909f8c7"
182
182
  }
@@ -25,6 +25,7 @@ const run_sst_command_js_1 = require("../sst-lifecycle/run-sst-command.js");
25
25
  const index_js_3 = require("../../workspace-deployment/index.js");
26
26
  const private_smoke_access_js_1 = require("../../foundation/private-smoke-access.js");
27
27
  const local_private_smoke_access_js_1 = require("../../foundation/local-private-smoke-access.js");
28
+ const capability_plan_js_1 = require("../../foundation/capability-plan.js");
28
29
  const sst_environment_source_js_1 = require("../sst-lifecycle/sst-environment-source.js");
29
30
  const SHA = /^[a-f0-9]{40,64}$/u;
30
31
  const ROLE = /^arn:(aws|aws-us-gov|aws-cn):iam::\d{12}:role\/[A-Za-z0-9+=,.@_/-]{1,512}$/u;
@@ -668,6 +669,30 @@ async function runNxTarget(input) {
668
669
  if (primaryFailure)
669
670
  throw primaryFailure.error;
670
671
  }
672
+ const FOUNDATION_DESCRIPTOR_PATH = 'packages/foundation/contracts/foundation.json';
673
+ /** The committed Foundation declaration is the only SQL signal; the generated
674
+ * Foundation publishes database discovery exactly when it declares postgres. */
675
+ function productionFoundationDeclaresPostgres(root) {
676
+ let descriptor;
677
+ try {
678
+ descriptor = JSON.parse((0, node_fs_1.readFileSync)((0, node_path_1.join)(root, FOUNDATION_DESCRIPTOR_PATH), 'utf8'));
679
+ }
680
+ catch {
681
+ throw new Error(`Production deployment requires the committed Foundation capability declaration ${FOUNDATION_DESCRIPTOR_PATH}.`);
682
+ }
683
+ return (0, capability_plan_js_1.planFoundationCapabilities)({ current: descriptor, requirements: [] })
684
+ .resources.postgres;
685
+ }
686
+ /** Declarations are monotonic: a SQL-free Foundation must never observe, gain,
687
+ * or lose a production database during a native deployment. The
688
+ * `priorDiscovery` check is a redundant second line: the preflight already
689
+ * rejects an existing database before any child deploys, so a prior `existing`
690
+ * state cannot reach the postflights today and no spec exercises that branch. */
691
+ function assertSqlFreeProductionDatabase(discovery, priorDiscovery) {
692
+ if (discovery.state !== 'absent' || priorDiscovery?.state === 'existing') {
693
+ throw new Error('Production Foundation database discovery is present although the Foundation does not declare postgres; deployment is blocked.');
694
+ }
695
+ }
671
696
  function productionConfirmation(account, region, commit) {
672
697
  return `deploy:production:${account}:${region}:${commit}`;
673
698
  }
@@ -728,6 +753,7 @@ async function runLocalDeploy(options, context, dependencies = {}) {
728
753
  let transport;
729
754
  let productionDatabaseBefore;
730
755
  let productionDatabaseAfterFoundation;
756
+ let productionPostgresDeclared = true;
731
757
  let outcome = 'failed';
732
758
  let failure;
733
759
  let finalPhase = 'validation';
@@ -803,6 +829,11 @@ async function runLocalDeploy(options, context, dependencies = {}) {
803
829
  if ((githubAuthority || options.stage !== 'development') && !source.clean) {
804
830
  throw new Error(`Local ${options.stage} deployment requires a completely clean source tree.`);
805
831
  }
832
+ if (options.stage === 'production') {
833
+ // Read only after the clean-source check so the declaration is bound to
834
+ // the digested tree, and before any AWS authority or transport exists.
835
+ productionPostgresDeclared = productionFoundationDeclaresPostgres(context.root);
836
+ }
806
837
  if (githubAuthority) {
807
838
  if (options.profile) {
808
839
  throw new Error('Protected GitHub deployment must use ambient OIDC credentials, not a local AWS profile.');
@@ -947,6 +978,11 @@ async function runLocalDeploy(options, context, dependencies = {}) {
947
978
  ...(transport ? { profile: transport.profile } : {}),
948
979
  requireExisting: false,
949
980
  });
981
+ // Fail before any mutation: deploying a SQL-free Foundation over an
982
+ // existing database would retire it outside the separate retirement path.
983
+ if (!productionPostgresDeclared) {
984
+ assertSqlFreeProductionDatabase(productionDatabaseBefore);
985
+ }
950
986
  }
951
987
  const localContainerRunId = localProtectedContainerProjects.length === 0
952
988
  ? ''
@@ -1122,17 +1158,24 @@ async function runLocalDeploy(options, context, dependencies = {}) {
1122
1158
  }
1123
1159
  : {}),
1124
1160
  expectedAccountId: account,
1125
- expectedCapacity: {
1126
- max: Number.parseFloat(profile.aurora.max),
1127
- min: Number.parseFloat(profile.aurora.min),
1128
- },
1161
+ ...(productionPostgresDeclared
1162
+ ? {
1163
+ expectedCapacity: {
1164
+ max: Number.parseFloat(profile.aurora.max),
1165
+ min: Number.parseFloat(profile.aurora.min),
1166
+ },
1167
+ }
1168
+ : {}),
1129
1169
  expectedRegion: region,
1130
1170
  expectedRoleArn,
1131
1171
  now: now(),
1132
1172
  priorDiscovery: productionDatabaseBefore,
1133
1173
  ...(transport ? { profile: transport.profile } : {}),
1134
- requireExisting: true,
1174
+ requireExisting: productionPostgresDeclared,
1135
1175
  });
1176
+ if (!productionPostgresDeclared) {
1177
+ assertSqlFreeProductionDatabase(productionDatabaseAfterFoundation, productionDatabaseBefore);
1178
+ }
1136
1179
  children.push({
1137
1180
  completedAt: now().toISOString(),
1138
1181
  outcome: 'succeeded',
@@ -1259,7 +1302,7 @@ async function runLocalDeploy(options, context, dependencies = {}) {
1259
1302
  const foundationProject = orderedProjects[0] ?? 'foundation';
1260
1303
  const profile = (0, index_js_2.productionFoundationProfileContract)(policy.productionFoundation.profile).contract;
1261
1304
  try {
1262
- await (dependencies.attestProductionDatabase ??
1305
+ const finalDiscovery = await (dependencies.attestProductionDatabase ??
1263
1306
  production_database_safeguard_js_1.attestProductionDatabaseLifecycle)({
1264
1307
  ...(boundaryEnvironment.AWS_CONFIG_FILE
1265
1308
  ? { configFilepath: boundaryEnvironment.AWS_CONFIG_FILE }
@@ -1270,17 +1313,24 @@ async function runLocalDeploy(options, context, dependencies = {}) {
1270
1313
  }
1271
1314
  : {}),
1272
1315
  expectedAccountId: account,
1273
- expectedCapacity: {
1274
- max: Number.parseFloat(profile.aurora.max),
1275
- min: Number.parseFloat(profile.aurora.min),
1276
- },
1316
+ ...(productionPostgresDeclared
1317
+ ? {
1318
+ expectedCapacity: {
1319
+ max: Number.parseFloat(profile.aurora.max),
1320
+ min: Number.parseFloat(profile.aurora.min),
1321
+ },
1322
+ }
1323
+ : {}),
1277
1324
  expectedRegion: region,
1278
1325
  expectedRoleArn,
1279
1326
  now: now(),
1280
1327
  priorDiscovery: productionDatabaseAfterFoundation,
1281
1328
  ...(transport ? { profile: transport.profile } : {}),
1282
- requireExisting: true,
1329
+ requireExisting: productionPostgresDeclared,
1283
1330
  });
1331
+ if (!productionPostgresDeclared) {
1332
+ assertSqlFreeProductionDatabase(finalDiscovery, productionDatabaseAfterFoundation);
1333
+ }
1284
1334
  children.push({
1285
1335
  completedAt: now().toISOString(),
1286
1336
  outcome: 'succeeded',
@@ -51,7 +51,7 @@ const SST_PROCESS_GROUP_POLL_MS = 25;
51
51
  const GOVERNED_PROCESS_GROUP_ENV = 'EBK_GOVERNED_PROCESS_GROUP';
52
52
  function assertGovernedAwsLifecyclePlatform(platform = process.platform) {
53
53
  if (platform === 'win32') {
54
- throw new Error('Governed AWS infrastructure lifecycle is unavailable on native Windows; use macOS, Linux, or WSL2. Native Windows remains supported for local development and verification.');
54
+ throw new Error('Governed AWS infrastructure lifecycle is unavailable on native Windows; use macOS, Linux, or WSL2. Native-Windows task Fleet is not validated in this prerelease either.');
55
55
  }
56
56
  }
57
57
  function isMissingProcessError(error) {
@@ -128,8 +128,9 @@ configuration. It is advisory, not proof that image authentication will fail:
128
128
  some valid helpers do not implement list. If Docker's actual image pull fails,
129
129
  repair/unlock that registry's selected helper; do not
130
130
  remove host credentials, drop CLI plugins or prune shared caches as a workaround.
131
- Docker is required only for declared container/local-SQL dependencies, not
132
- ordinary unit tests or DynamoDB feature editing.
131
+ Docker is required for Slice \`verify\` (its \`test-integration\` runs DynamoDB
132
+ Local, plus PostgreSQL when SQL is declared) and for declared container
133
+ dependencies, not for ordinary unit tests or feature editing.
133
134
 
134
135
  \`\`\`bash
135
136
  ebk dev up billing
@@ -143,8 +144,9 @@ owned fleet origin and browser-scoped trust; do not guess ports or alter host
143
144
  certificate trust. Keep the persisted opaque task identity on restart. Inspect
144
145
  owned status after interruption; use supported task recovery, not lock deletion.
145
146
 
146
- Use affected unit/type/lint/ownership checks during edits. Require Docker only
147
- for local container dependencies or external Blueprint execution; require AWS
147
+ Use affected unit/type/lint/ownership checks during edits. Require Docker for
148
+ Slice integration tests (\`test-integration\`, part of \`verify\`), local container
149
+ dependencies, or external Blueprint execution; require AWS
148
150
  credentials for actual task resources. Install browser binaries for browser
149
151
  proof. A Standard Slice includes a Python infrastructure provisioner even when
150
152
  its feature handlers use TypeScript: install uv ${tool_versions_js_1.WORKSPACE_UV_VERSION} (Python 3.12 or
@@ -29,7 +29,9 @@ ${tool_versions_js_1.WORKSPACE_UV_VERSION} (Python 3.12 or newer) on PATH before
29
29
  \`ebk deploy\`. Each Standard Slice stops before its SST child starts when
30
30
  \`uv --version\` fails; a full \`ebk deploy\` may already have deployed
31
31
  Foundation.
32
- Source-only feature checks do not require it.
32
+ Source-only feature checks do not require it. Slice \`verify\` runs
33
+ \`test-integration\`, which needs a running Docker engine for DynamoDB Local (and
34
+ PostgreSQL when SQL is declared).
33
35
  `,
34
36
  'docs/git-workflow.md': `# Protected Git workflow
35
37
 
@@ -163,10 +165,11 @@ committed policy and gives SST an invocation-private temporary profile rather
163
165
  than the operator's Identity Center profile chain.
164
166
 
165
167
  Governed AWS infrastructure planning and mutation are supported on macOS,
166
- Linux, and WSL2. Native Windows remains supported for local development and
167
- verification, but run \`deploy-local\`, \`infrastructure-plan\`, \`refresh\`,
168
- \`remove\`, and \`unlock\` from WSL2 or a macOS/Linux host. Generated protected
169
- GitHub AWS jobs continue to run on Ubuntu.
168
+ Linux, and WSL2. Native Windows can run local verification, but native-Windows
169
+ task Fleet (\`ebk dev\`) is not validated in this prerelease; use WSL2 or a
170
+ macOS/Linux host for Fleet development and for \`deploy-local\`,
171
+ \`infrastructure-plan\`, \`refresh\`, \`remove\`, and \`unlock\`. Generated
172
+ protected GitHub AWS jobs continue to run on Ubuntu.
170
173
 
171
174
  Development may deploy a dirty working tree without a prompt. Its receipt binds
172
175
  the exact HEAD and working-tree digest. Local staging requires a clean commit
@@ -669,8 +669,11 @@ export const expireHoldHandler = createContractHttpJsonHandler<
669
669
  `,
670
670
  'database/runtime-client.ts': `import {
671
671
  databaseTlsOptions,
672
+ isDatabaseConnectFailure,
672
673
  runWithDatabaseCredential,
673
674
  toNodePostgresConnectionOptions,
675
+ withDatabaseClient,
676
+ type DatabaseConnectRetry,
674
677
  type LinkedDatabaseResource,
675
678
  } from '@empire-builder-kit/runtime/database';
676
679
  import { createLogger } from '@empire-builder-kit/runtime/logging';
@@ -729,22 +732,72 @@ function linkedDatabase(): LinkedDatabaseResource {
729
732
  return resource as LinkedDatabaseResource;
730
733
  }
731
734
 
735
+ type PrimaryDatabaseWorkload =
736
+ | 'api-database-health'
737
+ | 'api-database-records'
738
+ | 'api-publish-created'
739
+ | 'async-inbox-cleanup'
740
+ | 'async-outbox-relay'
741
+ | 'async-provider-worker'
742
+ | 'data-export-cleanup'
743
+ | 'data-retention-job'
744
+ | 'event-consumer'
745
+ | 'web-server';
746
+
747
+ // Aurora Serverless v2 resume can take 15-30 s or longer. Only connect()
748
+ // retries, within a per-call wall-clock budget that leaves room for the
749
+ // operation: HTTP routes and the web server run on SST's 20 s default
750
+ // timeout, workers and the outbox relay on 1 minute, and jobs on 5 minutes.
751
+ // HTTP attempts are short so a held or slow resume connect is retried
752
+ // inside the budget instead of consuming it in a single attempt.
753
+ const connectPolicy: Record<
754
+ PrimaryDatabaseWorkload,
755
+ { attemptTimeoutMs: number; budgetMs: number }
756
+ > = {
757
+ 'api-database-health': { attemptTimeoutMs: 5_000, budgetMs: 15_000 },
758
+ 'api-database-records': { attemptTimeoutMs: 5_000, budgetMs: 15_000 },
759
+ 'api-publish-created': { attemptTimeoutMs: 5_000, budgetMs: 15_000 },
760
+ 'async-inbox-cleanup': { attemptTimeoutMs: 15_000, budgetMs: 120_000 },
761
+ 'async-outbox-relay': { attemptTimeoutMs: 15_000, budgetMs: 40_000 },
762
+ 'async-provider-worker': { attemptTimeoutMs: 15_000, budgetMs: 40_000 },
763
+ 'data-export-cleanup': { attemptTimeoutMs: 15_000, budgetMs: 120_000 },
764
+ 'data-retention-job': { attemptTimeoutMs: 15_000, budgetMs: 120_000 },
765
+ 'event-consumer': { attemptTimeoutMs: 15_000, budgetMs: 40_000 },
766
+ 'web-server': { attemptTimeoutMs: 5_000, budgetMs: 15_000 },
767
+ };
768
+
769
+ function logConnectionRetry(
770
+ context: { correlationId: string },
771
+ workload: PrimaryDatabaseWorkload,
772
+ retry: DatabaseConnectRetry,
773
+ ): void {
774
+ createLogger({
775
+ component: workload,
776
+ feature: 'database',
777
+ operation: 'database.connect',
778
+ slice: '${name}',
779
+ stage: process.env.EBK_STAGE ?? '',
780
+ workspace: '${workspace}',
781
+ }).warn('Retrying database connection', {
782
+ attributes: {
783
+ databaseEvent: 'database_runtime_connection_retry',
784
+ ...retry,
785
+ },
786
+ context: { correlationId: context.correlationId },
787
+ outcome: 'unknown',
788
+ });
789
+ }
790
+
732
791
  export async function withPrimaryDatabase<Result>(
733
792
  context: { correlationId: string },
734
- workload:
735
- | 'api-database-health'
736
- | 'api-database-records'
737
- | 'api-publish-created'
738
- | 'async-inbox-cleanup'
739
- | 'async-outbox-relay'
740
- | 'async-provider-worker'
741
- | 'data-export-cleanup'
742
- | 'data-retention-job'
743
- | 'event-consumer'
744
- | 'web-server',
793
+ workload: PrimaryDatabaseWorkload,
745
794
  operation: (client: PgClient) => Promise<Result>,
746
795
  ): Promise<Result> {
747
796
  const resource = linkedDatabase();
797
+ const policy = connectPolicy[workload];
798
+ // One per-call deadline spans credential resolution and any 28P01 refresh.
799
+ // Handlers that make several sequential calls bound their own total work.
800
+ const deadline = Date.now() + policy.budgetMs;
748
801
  return runWithDatabaseCredential(
749
802
  resource,
750
803
  resource.provider === 'aws-secrets-manager'
@@ -754,22 +807,25 @@ export async function withPrimaryDatabase<Result>(
754
807
  resolver: getSecretResolver(workload),
755
808
  }
756
809
  : undefined,
757
- async (credential) => {
758
- const client = new Client({
759
- ...toNodePostgresConnectionOptions(credential),
760
- connectionTimeoutMillis: 10_000,
761
- ssl: databaseTlsOptions(resource),
762
- });
763
- try {
764
- await client.connect();
765
- return await operation(client);
766
- } finally {
767
- await client.end().catch(() => undefined);
768
- }
769
- },
810
+ (credential) =>
811
+ withDatabaseClient(
812
+ {
813
+ attemptTimeoutMs: policy.attemptTimeoutMs,
814
+ createClient: (connectionTimeoutMillis) =>
815
+ new Client({
816
+ ...toNodePostgresConnectionOptions(credential),
817
+ connectionTimeoutMillis,
818
+ ssl: databaseTlsOptions(resource),
819
+ }),
820
+ deadline,
821
+ onRetry: (retry) => logConnectionRetry(context, workload, retry),
822
+ },
823
+ operation,
824
+ ),
825
+ // Refresh credentials only when connect() itself failed authentication,
826
+ // so a started operation is never run a second time.
770
827
  (error) =>
771
- error !== null &&
772
- typeof error === 'object' &&
828
+ isDatabaseConnectFailure(error) &&
773
829
  'code' in error &&
774
830
  error.code === '28P01',
775
831
  );
@@ -6045,6 +6045,55 @@ capacity/idempotency/provider-effect evidence, stop conditions, and the forward
6045
6045
  compensation plan. Archive-wide observed size is evidence, not an immutable
6046
6046
  range estimate, because the canonical archive continues receiving events while
6047
6047
  approval is pending.
6048
+
6049
+ ## Event delivery alarms
6050
+
6051
+ Domain events are delivered by the key-value stream Function and recovered by
6052
+ the scheduled key-value reconciliation Function. The \`async/outbox-relay\`
6053
+ Function drains only the isolated performance sink; \`outbox-relay-errors\`
6054
+ does not describe domain delivery. These alarms observe delivery only. No alarm
6055
+ pauses, drops, quarantines, caps, or retries an event intent, and reconciliation
6056
+ does not fail because some intents remain undelivered.
6057
+
6058
+ - \`key-value-stream-delivery-errors\` reports a failed invocation of the
6059
+ stream delivery Function: the stream identity or record shape was rejected,
6060
+ or the handler crashed. Per-record publish failures are partial batch
6061
+ failures, not Lambda errors. They surface through
6062
+ \`key-value-stream-iterator-age\` while retries hold the shard back, and
6063
+ through \`outbox-backlog\` and \`event-delivery-attempts\` after the mapping
6064
+ discards them (after three retries or one hour).
6065
+ - \`key-value-reconciliation-errors\` reports Lambda invocation errors for the
6066
+ exact reconciliation Function.
6067
+ - \`key-value-stream-iterator-age\` reports a stream iterator at least ten
6068
+ minutes old. The mapping discards records older than one hour; their retained
6069
+ intents then wait for reconciliation.
6070
+ - \`outbox-backlog\` reports \`OutboxBacklog\`: intents a reconciliation run
6071
+ attempted but could not deliver, plus rejected intents, in two consecutive
6072
+ hours. Each run reads at most one 25-intent page per shard, so the value is a
6073
+ lower bound rather than a global count. Intents leased by another delivery
6074
+ owner are in flight and are not counted.
6075
+ - \`event-intent-rejected\` reports retained intents that fail validation.
6076
+ They are never dispatched or deleted, so the alarm recurs until an owner
6077
+ investigates the content-free \`ebk-key-value-invalid-intents\` record.
6078
+ - \`event-delivery-attempts\` reports \`EventDeliveryMaxAttempts\` of at least
6079
+ ten delivery leases on one due intent. Attempts are reported, never enforced.
6080
+
6081
+ SQL Slices only (see docs/runbooks/sql-events.md):
6082
+
6083
+ - \`sql-outbox-relay-errors\` reports Lambda errors for the exact SQL outbox
6084
+ relay Function, which fails every run that retains a failed delivery.
6085
+ - \`sql-outbox-backlog\` reports the Minimum of \`SqlOutboxBacklog\`, the
6086
+ global unpublished-intent count, over two consecutive hours (3600 s x 2): it
6087
+ alarms only when no relay run in either hour drained the outbox. A paused
6088
+ relay reports nothing, so the alarm is absent rather than breaching while
6089
+ paused.
6090
+ - \`sql-outbox-attempts\` reports \`SqlOutboxMaxAttempts\` of at least ten
6091
+ relay attempts on one unpublished intent. It is observational; no intent is
6092
+ capped or dropped.
6093
+
6094
+ Diagnose with the content-free \`ebk-key-value-reconciliation\` log summary and
6095
+ the EventBridge target before any recovery. Do not delete or edit retained
6096
+ intents to clear an alarm.
6048
6097
  `,
6049
6098
  };
6050
6099
  }
@@ -773,6 +773,142 @@ ${` const keyValue = createKeyValueInfrastructure({
773
773
  lambdaLogging,
774
774
  });
775
775
  `}\
776
+ // Domain delivery is the key-value stream plus scheduled reconciliation; the
777
+ // async/outbox-relay Function drains only the isolated performance sink.
778
+ // Observation only: alarms never pause, drop, or retry an event intent.
779
+ const keyValueStreamDeliveryErrorAlarm = new aws.cloudwatch.MetricAlarm(
780
+ '${pascalName}KeyValueStreamDeliveryErrors',
781
+ {
782
+ alarmActions: [foundation.discovery.alertTopicArn],
783
+ alarmDescription: operationalAlarmDescription({
784
+ businessSymptom:
785
+ 'The key-value stream delivery Function failed an invocation.',
786
+ id: 'key-value-stream-delivery-errors',
787
+ runbook: 'docs/runbooks/async-recovery.md',
788
+ }),
789
+ comparisonOperator: 'GreaterThanOrEqualToThreshold',
790
+ dimensions: { FunctionName: keyValue.dispatcher.name },
791
+ evaluationPeriods: 1,
792
+ metricName: 'Errors',
793
+ name: '${name}-' + $app.stage + '-key-value-stream-delivery-errors',
794
+ namespace: 'AWS/Lambda',
795
+ period: 300,
796
+ statistic: 'Sum',
797
+ threshold: 1,
798
+ treatMissingData: 'notBreaching',
799
+ },
800
+ );
801
+ const keyValueStreamIteratorAgeAlarm = new aws.cloudwatch.MetricAlarm(
802
+ '${pascalName}KeyValueStreamIteratorAge',
803
+ {
804
+ alarmActions: [foundation.discovery.alertTopicArn],
805
+ alarmDescription: operationalAlarmDescription({
806
+ businessSymptom:
807
+ 'Prompt domain-event delivery from the key-value stream is falling behind.',
808
+ id: 'key-value-stream-iterator-age',
809
+ runbook: 'docs/runbooks/async-recovery.md',
810
+ }),
811
+ comparisonOperator: 'GreaterThanOrEqualToThreshold',
812
+ dimensions: { FunctionName: keyValue.dispatcher.name },
813
+ evaluationPeriods: 1,
814
+ metricName: 'IteratorAge',
815
+ name: '${name}-' + $app.stage + '-key-value-stream-iterator-age',
816
+ namespace: 'AWS/Lambda',
817
+ period: 300,
818
+ statistic: 'Maximum',
819
+ threshold: 600_000,
820
+ treatMissingData: 'notBreaching',
821
+ },
822
+ );
823
+ const keyValueReconciliationErrorAlarm = new aws.cloudwatch.MetricAlarm(
824
+ '${pascalName}KeyValueReconciliationErrors',
825
+ {
826
+ alarmActions: [foundation.discovery.alertTopicArn],
827
+ alarmDescription: operationalAlarmDescription({
828
+ businessSymptom: 'Durable domain-event recovery is failing.',
829
+ id: 'key-value-reconciliation-errors',
830
+ runbook: 'docs/runbooks/async-recovery.md',
831
+ }),
832
+ comparisonOperator: 'GreaterThanOrEqualToThreshold',
833
+ dimensions: { FunctionName: keyValue.reconciler.name },
834
+ evaluationPeriods: 1,
835
+ metricName: 'Errors',
836
+ name: '${name}-' + $app.stage + '-key-value-reconciliation-errors',
837
+ namespace: 'AWS/Lambda',
838
+ period: 300,
839
+ statistic: 'Sum',
840
+ threshold: 1,
841
+ treatMissingData: 'notBreaching',
842
+ },
843
+ );
844
+ // Reconciliation gauges are bounded per-run samples (one page per shard),
845
+ // so hourly periods cover the default and any sub-hourly reconciliation
846
+ // cadence. A slower configured cadence leaves empty periods, so these alarms
847
+ // can miss a backlog (see each registration's expectedFalseNegativeRisk).
848
+ const outboxBacklogAlarm = new aws.cloudwatch.MetricAlarm(
849
+ '${pascalName}OutboxBacklog',
850
+ {
851
+ alarmActions: [foundation.discovery.alertTopicArn],
852
+ alarmDescription: operationalAlarmDescription({
853
+ businessSymptom: 'Durable domain-event intents remain undelivered.',
854
+ id: 'outbox-backlog',
855
+ runbook: 'docs/runbooks/async-recovery.md',
856
+ }),
857
+ comparisonOperator: 'GreaterThanOrEqualToThreshold',
858
+ dimensions: { Slice: '${name}', Stage: $app.stage },
859
+ evaluationPeriods: 2,
860
+ metricName: 'OutboxBacklog',
861
+ name: '${name}-' + $app.stage + '-outbox-backlog',
862
+ namespace: 'EBK/Async',
863
+ period: 3600,
864
+ statistic: 'Maximum',
865
+ threshold: 1,
866
+ treatMissingData: 'notBreaching',
867
+ },
868
+ );
869
+ const eventIntentRejectedAlarm = new aws.cloudwatch.MetricAlarm(
870
+ '${pascalName}EventIntentRejected',
871
+ {
872
+ alarmActions: [foundation.discovery.alertTopicArn],
873
+ alarmDescription: operationalAlarmDescription({
874
+ businessSymptom: 'A retained domain-event intent cannot be delivered.',
875
+ id: 'event-intent-rejected',
876
+ runbook: 'docs/runbooks/async-recovery.md',
877
+ }),
878
+ comparisonOperator: 'GreaterThanOrEqualToThreshold',
879
+ dimensions: { Slice: '${name}', Stage: $app.stage },
880
+ evaluationPeriods: 1,
881
+ metricName: 'EventReconciliationRejected',
882
+ name: '${name}-' + $app.stage + '-event-intent-rejected',
883
+ namespace: 'EBK/Async',
884
+ period: 3600,
885
+ statistic: 'Maximum',
886
+ threshold: 1,
887
+ treatMissingData: 'notBreaching',
888
+ },
889
+ );
890
+ const eventDeliveryAttemptsAlarm = new aws.cloudwatch.MetricAlarm(
891
+ '${pascalName}EventDeliveryAttempts',
892
+ {
893
+ alarmActions: [foundation.discovery.alertTopicArn],
894
+ alarmDescription: operationalAlarmDescription({
895
+ businessSymptom:
896
+ 'A domain-event intent is repeatedly failing delivery.',
897
+ id: 'event-delivery-attempts',
898
+ runbook: 'docs/runbooks/async-recovery.md',
899
+ }),
900
+ comparisonOperator: 'GreaterThanOrEqualToThreshold',
901
+ dimensions: { Slice: '${name}', Stage: $app.stage },
902
+ evaluationPeriods: 1,
903
+ metricName: 'EventDeliveryMaxAttempts',
904
+ name: '${name}-' + $app.stage + '-event-delivery-attempts',
905
+ namespace: 'EBK/Async',
906
+ period: 3600,
907
+ statistic: 'Maximum',
908
+ threshold: 10,
909
+ treatMissingData: 'notBreaching',
910
+ },
911
+ );
776
912
  const nativeAuthority = createStandardSliceNativeRuntimeAuthority({
777
913
  ${sql
778
914
  ? ` clusterId: foundation.discovery.clusterId,
@@ -1840,6 +1976,72 @@ ${sql
1840
1976
  managedLogGroup, lambdaLogging,
1841
1977
  ${''}\
1842
1978
  });
1979
+ const sqlOutboxRelayErrorAlarm = new aws.cloudwatch.MetricAlarm(
1980
+ '${pascalName}SqlOutboxRelayErrors',
1981
+ {
1982
+ alarmActions: [foundation.discovery.alertTopicArn],
1983
+ alarmDescription: operationalAlarmDescription({
1984
+ businessSymptom: 'Relational event delivery is failing.',
1985
+ id: 'sql-outbox-relay-errors',
1986
+ runbook: 'docs/runbooks/async-recovery.md',
1987
+ }),
1988
+ comparisonOperator: 'GreaterThanOrEqualToThreshold',
1989
+ dimensions: { FunctionName: sqlOutbox.relay.name },
1990
+ evaluationPeriods: 1,
1991
+ metricName: 'Errors',
1992
+ name: '${name}-' + $app.stage + '-sql-outbox-relay-errors',
1993
+ namespace: 'AWS/Lambda',
1994
+ period: 300,
1995
+ statistic: 'Sum',
1996
+ threshold: 1,
1997
+ treatMissingData: 'notBreaching',
1998
+ },
1999
+ );
2000
+ // The relay counts every unpublished intent; Minimum ignores healthy
2001
+ // in-flight batches and alarms only when no run in an hour drained it.
2002
+ const sqlOutboxBacklogAlarm = new aws.cloudwatch.MetricAlarm(
2003
+ '${pascalName}SqlOutboxBacklog',
2004
+ {
2005
+ alarmActions: [foundation.discovery.alertTopicArn],
2006
+ alarmDescription: operationalAlarmDescription({
2007
+ businessSymptom: 'Relational event intents remain undelivered.',
2008
+ id: 'sql-outbox-backlog',
2009
+ runbook: 'docs/runbooks/async-recovery.md',
2010
+ }),
2011
+ comparisonOperator: 'GreaterThanOrEqualToThreshold',
2012
+ dimensions: { Slice: '${name}', Stage: $app.stage },
2013
+ evaluationPeriods: 2,
2014
+ metricName: 'SqlOutboxBacklog',
2015
+ name: '${name}-' + $app.stage + '-sql-outbox-backlog',
2016
+ namespace: 'EBK/Async',
2017
+ period: 3600,
2018
+ statistic: 'Minimum',
2019
+ threshold: 1,
2020
+ treatMissingData: 'notBreaching',
2021
+ },
2022
+ );
2023
+ const sqlOutboxAttemptsAlarm = new aws.cloudwatch.MetricAlarm(
2024
+ '${pascalName}SqlOutboxAttempts',
2025
+ {
2026
+ alarmActions: [foundation.discovery.alertTopicArn],
2027
+ alarmDescription: operationalAlarmDescription({
2028
+ businessSymptom:
2029
+ 'A relational event intent is repeatedly failing delivery.',
2030
+ id: 'sql-outbox-attempts',
2031
+ runbook: 'docs/runbooks/async-recovery.md',
2032
+ }),
2033
+ comparisonOperator: 'GreaterThanOrEqualToThreshold',
2034
+ dimensions: { Slice: '${name}', Stage: $app.stage },
2035
+ evaluationPeriods: 1,
2036
+ metricName: 'SqlOutboxMaxAttempts',
2037
+ name: '${name}-' + $app.stage + '-sql-outbox-attempts',
2038
+ namespace: 'EBK/Async',
2039
+ period: 3600,
2040
+ statistic: 'Maximum',
2041
+ threshold: 10,
2042
+ treatMissingData: 'notBreaching',
2043
+ },
2044
+ );
1843
2045
  `
1844
2046
  : ''}\
1845
2047
  const asyncInboxCleanup = new sst.aws.Function(
@@ -2749,7 +2951,8 @@ ${''}\
2749
2951
  {
2750
2952
  alarmActions: [foundation.discovery.alertTopicArn],
2751
2953
  alarmDescription: operationalAlarmDescription({
2752
- businessSymptom: 'The transactional outbox relay is failing.',
2954
+ businessSymptom:
2955
+ 'The isolated performance-sink outbox relay is failing.',
2753
2956
  id: 'outbox-relay-errors',
2754
2957
  runbook: 'docs/runbooks/async-recovery.md',
2755
2958
  }),
@@ -2834,27 +3037,6 @@ ${''}\
2834
3037
  treatMissingData: 'breaching',
2835
3038
  },
2836
3039
  );
2837
- const outboxBacklogAlarm = new aws.cloudwatch.MetricAlarm(
2838
- '${pascalName}OutboxBacklog',
2839
- {
2840
- alarmActions: [foundation.discovery.alertTopicArn],
2841
- alarmDescription: operationalAlarmDescription({
2842
- businessSymptom: 'Transactional outbox work is accumulating.',
2843
- id: 'outbox-backlog',
2844
- runbook: 'docs/runbooks/async-recovery.md',
2845
- }),
2846
- comparisonOperator: 'GreaterThanOrEqualToThreshold',
2847
- dimensions: { Slice: '${name}', Stage: $app.stage },
2848
- evaluationPeriods: 2,
2849
- metricName: 'OutboxBacklog',
2850
- name: '${name}-' + $app.stage + '-outbox-backlog',
2851
- namespace: 'EBK/Async',
2852
- period: 300,
2853
- statistic: 'Maximum',
2854
- threshold: 1,
2855
- treatMissingData: 'notBreaching',
2856
- },
2857
- );
2858
3040
  const inboxCleanupBacklogAlarm = new aws.cloudwatch.MetricAlarm(
2859
3041
  '${pascalName}InboxCleanupBacklog',
2860
3042
  {
@@ -3000,7 +3182,7 @@ ${''}\
3000
3182
  performanceLoadCleanupBacklogAlarm,
3001
3183
  performanceLoadCleanupDeadlineAlarm,
3002
3184
  asyncOutboxRelay,
3003
- ${sql ? ' sqlOutbox,\n' : ''}\
3185
+ ${sql ? ' sqlOutbox,\n sqlOutboxAttemptsAlarm,\n sqlOutboxBacklogAlarm,\n sqlOutboxRelayErrorAlarm,\n' : ''}\
3004
3186
  asyncRecoveryLedger,
3005
3187
  asyncScheduleTarget,
3006
3188
  asyncTargetDlq,
@@ -3027,6 +3209,11 @@ ${' keyValue,\n'}\
3027
3209
  eventQueue,
3028
3210
  machineClientProvisioner,
3029
3211
  migrationControls,
3212
+ eventDeliveryAttemptsAlarm,
3213
+ eventIntentRejectedAlarm,
3214
+ keyValueReconciliationErrorAlarm,
3215
+ keyValueStreamDeliveryErrorAlarm,
3216
+ keyValueStreamIteratorAgeAlarm,
3030
3217
  outboxBacklogAlarm,
3031
3218
  outboxRelayAlarm,
3032
3219
  outboxRelayRule,
@@ -5,11 +5,13 @@ exports.renderKeyValueDeliverySource = renderKeyValueDeliverySource;
5
5
  function renderKeyValueDeliverySource(name) {
6
6
  return `import { EventBridgeClient, PutEventsCommand } from '@aws-sdk/client-eventbridge';
7
7
  import { DYNAMO_OUTBOX_SHARDS } from '@empire-builder-kit/runtime/key-value';
8
+ import { createMetricEmitter } from '@empire-builder-kit/runtime/observability';
8
9
  import type { PublishableEventEntry } from '@empire-builder-kit/runtime/events';
9
10
  import { dataStore } from '../data/key-value.js';
10
11
  import { reconcileRecordDeletions } from '../data/record-deletion.js';
11
12
 
12
13
  const client = new EventBridgeClient({ maxAttempts: 2 });
14
+ const metrics = createMetricEmitter();
13
15
  type Store = ReturnType<typeof dataStore>;
14
16
  type Cursor = NonNullable<Awaited<ReturnType<Store['listPending']>>['cursor']>;
15
17
  // Only the provider fields consumed below are part of this handler's input.
@@ -77,9 +79,16 @@ export async function stream(event: DynamoDBStreamEvent) {
77
79
  return { batchItemFailures };
78
80
  }
79
81
 
82
+ // EMF gauges are observational. A lost metric returns undefined and never
83
+ // changes delivery, checkpoints, or the reconciliation outcome.
84
+ function observe(metricName: string, value: number) {
85
+ metrics.emit({ namespace: 'EBK/Async', metricName, unit: 'Count', value,
86
+ dimensions: { slice: '${name}', stage: process.env.EBK_STAGE ?? 'local' } });
87
+ }
88
+
80
89
  export async function reconcile() {
81
90
  const store = dataStore();
82
- const summary = { kind: 'ebk-key-value-reconciliation', delivered: 0, deliveryPending: 0, retryRequired: 0, rejected: 0, shards: 0 };
91
+ const summary = { kind: 'ebk-key-value-reconciliation', delivered: 0, deliveryPending: 0, retryRequired: 0, rejected: 0, maxAttempts: 0, shards: 0 };
83
92
  // One page per shard per run bounds work and prevents a poison-filled first
84
93
  // page from starving the remainder forever. Wrap after the final page.
85
94
  for (let shard = 0; shard < DYNAMO_OUTBOX_SHARDS; shard++) {
@@ -94,6 +103,7 @@ export async function reconcile() {
94
103
  kind: 'ebk-key-value-invalid-intents', shard, records: page.rejected,
95
104
  }));
96
105
  for (const intent of page.events) {
106
+ summary.maxAttempts = Math.max(summary.maxAttempts, intent.attempts);
97
107
  try {
98
108
  const result = await store.dispatch(intent.id, publish);
99
109
  if (delivered(result)) summary.delivered++;
@@ -110,6 +120,13 @@ export async function reconcile() {
110
120
  }] });
111
121
  summary.shards++;
112
122
  }
123
+ // One page per shard is a bounded sample, not a global count: each gauge is a
124
+ // lower bound for this run. Intents leased by another owner are in flight and
125
+ // excluded; a dead owner's intent is due again after its lease expires.
126
+ observe('OutboxBacklog', summary.retryRequired + summary.rejected);
127
+ observe('EventReconciliationRetryRequired', summary.retryRequired);
128
+ observe('EventReconciliationRejected', summary.rejected);
129
+ observe('EventDeliveryMaxAttempts', summary.maxAttempts);
113
130
  const deletionCompletion = await reconcileRecordDeletions();
114
131
  console.info(JSON.stringify({ ...summary, deletionCompletion }));
115
132
  return { ...summary, deletionCompletion,
@@ -59,7 +59,10 @@ Run \`ebk dev up ${name}\`. The framework reuses this worktree's persisted opaqu
59
59
  task identity, starts the selected Slice and declared dependencies, and runs
60
60
  SST live mode through the workspace-pinned Nx task. Native persistence uses the
61
61
  task's real AWS DynamoDB resources; Docker PostgreSQL starts only for declared
62
- relational workloads. A caller cannot supply or persist a stage override. The
62
+ relational workloads. \`verify\` runs \`test-integration\`, which starts DynamoDB
63
+ Local (and PostgreSQL when declared) in disposable Docker containers, so
64
+ verification needs a running Docker engine. A caller cannot supply or persist a
65
+ stage override. The
63
66
  fleet supervisor allocates the internal Next.js port and reports the HTTPS
64
67
  readiness URL. Development accepts a dirty tree and requires no production
65
68
  confirmation or release artifact.
@@ -1,2 +1,2 @@
1
1
  /** @internal */
2
- export declare function renderProductReliabilityFiles(name: string, pascalName: string, reviewedAt: string): Readonly<Record<string, string>>;
2
+ export declare function renderProductReliabilityFiles(name: string, pascalName: string, reviewedAt: string, sql?: boolean): Readonly<Record<string, string>>;
@@ -127,7 +127,7 @@ function reliabilityContract(name, reviewedAt) {
127
127
  slice: name,
128
128
  };
129
129
  }
130
- function alarmRegistrations(name) {
130
+ function alarmRegistrations(name, sql) {
131
131
  const owner = `${name}-product`;
132
132
  const route = `${name}-on-call`;
133
133
  const common = {
@@ -243,8 +243,8 @@ function alarmRegistrations(name) {
243
243
  },
244
244
  {
245
245
  ...common,
246
- businessSymptom: 'A generated Function has a terminal invocation error.',
247
- condition: 'AWS/Lambda reports one or more errors for the exact generated cleanup Function.',
246
+ businessSymptom: 'The generated confidential-export cleanup Function has a terminal invocation error.',
247
+ condition: 'AWS/Lambda reports one or more errors in one minute for the exact generated data-export cleanup Function. Event-delivery Functions have their own error alarms.',
248
248
  dependencyAlarmIds: [],
249
249
  id: 'function-failure',
250
250
  runbookRef: 'runbook:api-recovery',
@@ -306,8 +306,8 @@ function alarmRegistrations(name) {
306
306
  },
307
307
  {
308
308
  ...common,
309
- businessSymptom: 'The transactional outbox relay is failing.',
310
- condition: 'The generated outbox relay reports one or more errors.',
309
+ businessSymptom: 'The isolated performance-sink outbox relay is failing.',
310
+ condition: 'AWS/Lambda reports one or more errors in five minutes for the generated async/outbox-relay Function, which drains only the isolated performance sink. Domain-event delivery is the key-value stream and reconciliation.',
311
311
  dependencyAlarmIds: [],
312
312
  id: 'outbox-relay-errors',
313
313
  runbookRef: 'runbook:async-recovery',
@@ -324,13 +324,62 @@ function alarmRegistrations(name) {
324
324
  },
325
325
  {
326
326
  ...common,
327
- businessSymptom: 'Transactional outbox work is accumulating.',
328
- condition: 'The transactional outbox backlog remains nonempty.',
329
- dependencyAlarmIds: ['outbox-relay-errors'],
327
+ businessSymptom: 'The key-value stream delivery Function failed an invocation.',
328
+ condition: 'AWS/Lambda reports one or more errors in five minutes for the exact key-value stream delivery Function: the stream identity or record shape was rejected, or the handler crashed. Per-record publish failures are partial batch failures, not Lambda errors; they surface through key-value-stream-iterator-age while retries hold the shard back, and through outbox-backlog and event-delivery-attempts after the mapping discards them.',
329
+ dependencyAlarmIds: [],
330
+ id: 'key-value-stream-delivery-errors',
331
+ runbookRef: 'runbook:async-recovery',
332
+ severity: 'high',
333
+ },
334
+ {
335
+ ...common,
336
+ businessSymptom: 'Prompt domain-event delivery from the key-value stream is falling behind.',
337
+ condition: 'AWS/Lambda reports a stream iterator age of at least ten minutes for the exact key-value stream delivery Function. The mapping discards records older than one hour; reconciliation then recovers their retained intents.',
338
+ dependencyAlarmIds: ['key-value-stream-delivery-errors'],
339
+ id: 'key-value-stream-iterator-age',
340
+ runbookRef: 'runbook:async-recovery',
341
+ severity: 'high',
342
+ },
343
+ {
344
+ ...common,
345
+ businessSymptom: 'Durable domain-event recovery is failing.',
346
+ condition: 'AWS/Lambda reports one or more errors in five minutes for the exact key-value reconciliation Function. Partial runs do not error; the backlog, rejected-intent, and attempts alarms report them.',
347
+ dependencyAlarmIds: [],
348
+ id: 'key-value-reconciliation-errors',
349
+ runbookRef: 'runbook:async-recovery',
350
+ severity: 'high',
351
+ },
352
+ {
353
+ ...common,
354
+ businessSymptom: 'Durable domain-event intents remain undelivered.',
355
+ condition: 'In two consecutive hours, key-value reconciliation reported one or more due intents it could not deliver (retry-required plus rejected). Each run reads at most one 25-intent page per shard, so the value is a lower bound, not a global count, and excludes intents leased by another delivery owner.',
356
+ dependencyAlarmIds: [
357
+ 'key-value-reconciliation-errors',
358
+ 'key-value-stream-delivery-errors',
359
+ ],
360
+ expectedFalseNegativeRisk: 'Intents beyond the sampled pages, or behind a paused outbox, are not counted until a later run reaches them. When the configured reconciliation cadence is slower than one hour, some hourly periods have no data and the two-period condition may never be met.',
330
361
  id: 'outbox-backlog',
331
362
  runbookRef: 'runbook:async-recovery',
332
363
  severity: 'high',
333
364
  },
365
+ {
366
+ ...common,
367
+ businessSymptom: 'A retained domain-event intent cannot be delivered.',
368
+ condition: 'Within an hour, key-value reconciliation reported one or more retained intents that fail validation. Rejected intents are never dispatched or deleted, so this recurs until an operator investigates.',
369
+ dependencyAlarmIds: [],
370
+ id: 'event-intent-rejected',
371
+ runbookRef: 'runbook:async-recovery',
372
+ severity: 'high',
373
+ },
374
+ {
375
+ ...common,
376
+ businessSymptom: 'A domain-event intent is repeatedly failing delivery.',
377
+ condition: 'Within an hour, key-value reconciliation observed a due intent with at least ten delivery leases (attempts). Attempts are reported only; no intent is capped, dropped, or quarantined.',
378
+ dependencyAlarmIds: ['key-value-reconciliation-errors'],
379
+ id: 'event-delivery-attempts',
380
+ runbookRef: 'runbook:async-recovery',
381
+ severity: 'high',
382
+ },
334
383
  {
335
384
  ...common,
336
385
  businessSymptom: 'Terminal asynchronous inbox records are accumulating.',
@@ -340,6 +389,38 @@ function alarmRegistrations(name) {
340
389
  runbookRef: 'runbook:async-recovery',
341
390
  severity: 'high',
342
391
  },
392
+ ...(sql
393
+ ? [
394
+ {
395
+ ...common,
396
+ businessSymptom: 'Relational event delivery is failing.',
397
+ condition: 'AWS/Lambda reports one or more errors in five minutes for the exact SQL outbox relay Function, which fails every run that retains a failed delivery.',
398
+ dependencyAlarmIds: [],
399
+ id: 'sql-outbox-relay-errors',
400
+ runbookRef: 'runbook:async-recovery',
401
+ severity: 'high',
402
+ },
403
+ {
404
+ ...common,
405
+ businessSymptom: 'Relational event intents remain undelivered.',
406
+ condition: 'Every SQL relay run in two consecutive hours ended with one or more unpublished intents (Minimum of a global count). A paused relay reports nothing.',
407
+ dependencyAlarmIds: ['sql-outbox-relay-errors'],
408
+ expectedFalseNegativeRisk: 'A paused relay reports nothing. When the configured sqlOutboxRecovery cadence is slower than one hour and no notification wakes the relay, some hourly periods have no data and the two-period condition may never be met.',
409
+ id: 'sql-outbox-backlog',
410
+ runbookRef: 'runbook:async-recovery',
411
+ severity: 'high',
412
+ },
413
+ {
414
+ ...common,
415
+ businessSymptom: 'A relational event intent is repeatedly failing delivery.',
416
+ condition: 'Within an hour, an unpublished SQL intent reached at least ten relay attempts. Attempts are reported only; no intent is capped or dropped.',
417
+ dependencyAlarmIds: ['sql-outbox-relay-errors'],
418
+ id: 'sql-outbox-attempts',
419
+ runbookRef: 'runbook:async-recovery',
420
+ severity: 'high',
421
+ },
422
+ ]
423
+ : []),
343
424
  ];
344
425
  return registrations.map((registration) => (0, reliability_1.validateAlarmRegistrationMetadata)(registration));
345
426
  }
@@ -514,7 +595,7 @@ function reliabilityExercises(name) {
514
595
  },
515
596
  ];
516
597
  }
517
- function deployedAlarmResources(name) {
598
+ function deployedAlarmResources(name, sql) {
518
599
  const primary = {
519
600
  actions: 'primary-route',
520
601
  comparisonOperator: 'GreaterThanOrEqualToThreshold',
@@ -636,7 +717,7 @@ function deployedAlarmResources(name) {
636
717
  Slice: name,
637
718
  Stage: '{stage}',
638
719
  },
639
- evaluationPeriods: 1,
720
+ evaluationPeriods: 2,
640
721
  id: 'performance-load-cleanup-backlog',
641
722
  metricName: 'PerformanceLoadCleanupBacklog',
642
723
  name: `${name}-{stage}-performance-load-cleanup-backlog`,
@@ -839,6 +920,51 @@ function deployedAlarmResources(name) {
839
920
  threshold: 1,
840
921
  treatMissingData: 'notBreaching',
841
922
  },
923
+ {
924
+ ...primary,
925
+ dimensionNames: ['FunctionName'],
926
+ evaluationPeriods: 1,
927
+ id: 'key-value-stream-delivery-errors',
928
+ metricName: 'Errors',
929
+ name: `${name}-{stage}-key-value-stream-delivery-errors`,
930
+ namespace: 'AWS/Lambda',
931
+ period: 300,
932
+ runbook: 'docs/runbooks/async-recovery.md',
933
+ signal: 'metric',
934
+ statistic: 'Sum',
935
+ threshold: 1,
936
+ treatMissingData: 'notBreaching',
937
+ },
938
+ {
939
+ ...primary,
940
+ dimensionNames: ['FunctionName'],
941
+ evaluationPeriods: 1,
942
+ id: 'key-value-stream-iterator-age',
943
+ metricName: 'IteratorAge',
944
+ name: `${name}-{stage}-key-value-stream-iterator-age`,
945
+ namespace: 'AWS/Lambda',
946
+ period: 300,
947
+ runbook: 'docs/runbooks/async-recovery.md',
948
+ signal: 'metric',
949
+ statistic: 'Maximum',
950
+ threshold: 600_000,
951
+ treatMissingData: 'notBreaching',
952
+ },
953
+ {
954
+ ...primary,
955
+ dimensionNames: ['FunctionName'],
956
+ evaluationPeriods: 1,
957
+ id: 'key-value-reconciliation-errors',
958
+ metricName: 'Errors',
959
+ name: `${name}-{stage}-key-value-reconciliation-errors`,
960
+ namespace: 'AWS/Lambda',
961
+ period: 300,
962
+ runbook: 'docs/runbooks/async-recovery.md',
963
+ signal: 'metric',
964
+ statistic: 'Sum',
965
+ threshold: 1,
966
+ treatMissingData: 'notBreaching',
967
+ },
842
968
  {
843
969
  ...primary,
844
970
  dimensionNames: ['Slice', 'Stage'],
@@ -848,13 +974,96 @@ function deployedAlarmResources(name) {
848
974
  metricName: 'OutboxBacklog',
849
975
  name: `${name}-{stage}-outbox-backlog`,
850
976
  namespace: 'EBK/Async',
851
- period: 300,
977
+ period: 3600,
852
978
  runbook: 'docs/runbooks/async-recovery.md',
853
979
  signal: 'metric',
854
980
  statistic: 'Maximum',
855
981
  threshold: 1,
856
982
  treatMissingData: 'notBreaching',
857
983
  },
984
+ {
985
+ ...primary,
986
+ dimensionNames: ['Slice', 'Stage'],
987
+ dimensionValues: { Slice: name, Stage: '{stage}' },
988
+ evaluationPeriods: 1,
989
+ id: 'event-intent-rejected',
990
+ metricName: 'EventReconciliationRejected',
991
+ name: `${name}-{stage}-event-intent-rejected`,
992
+ namespace: 'EBK/Async',
993
+ period: 3600,
994
+ runbook: 'docs/runbooks/async-recovery.md',
995
+ signal: 'metric',
996
+ statistic: 'Maximum',
997
+ threshold: 1,
998
+ treatMissingData: 'notBreaching',
999
+ },
1000
+ {
1001
+ ...primary,
1002
+ dimensionNames: ['Slice', 'Stage'],
1003
+ dimensionValues: { Slice: name, Stage: '{stage}' },
1004
+ evaluationPeriods: 1,
1005
+ id: 'event-delivery-attempts',
1006
+ metricName: 'EventDeliveryMaxAttempts',
1007
+ name: `${name}-{stage}-event-delivery-attempts`,
1008
+ namespace: 'EBK/Async',
1009
+ period: 3600,
1010
+ runbook: 'docs/runbooks/async-recovery.md',
1011
+ signal: 'metric',
1012
+ statistic: 'Maximum',
1013
+ threshold: 10,
1014
+ treatMissingData: 'notBreaching',
1015
+ },
1016
+ ...(sql
1017
+ ? [
1018
+ {
1019
+ ...primary,
1020
+ dimensionNames: ['FunctionName'],
1021
+ evaluationPeriods: 1,
1022
+ id: 'sql-outbox-relay-errors',
1023
+ metricName: 'Errors',
1024
+ name: `${name}-{stage}-sql-outbox-relay-errors`,
1025
+ namespace: 'AWS/Lambda',
1026
+ period: 300,
1027
+ runbook: 'docs/runbooks/async-recovery.md',
1028
+ signal: 'metric',
1029
+ statistic: 'Sum',
1030
+ threshold: 1,
1031
+ treatMissingData: 'notBreaching',
1032
+ },
1033
+ {
1034
+ ...primary,
1035
+ dimensionNames: ['Slice', 'Stage'],
1036
+ dimensionValues: { Slice: name, Stage: '{stage}' },
1037
+ evaluationPeriods: 2,
1038
+ id: 'sql-outbox-backlog',
1039
+ metricName: 'SqlOutboxBacklog',
1040
+ name: `${name}-{stage}-sql-outbox-backlog`,
1041
+ namespace: 'EBK/Async',
1042
+ period: 3600,
1043
+ runbook: 'docs/runbooks/async-recovery.md',
1044
+ signal: 'metric',
1045
+ statistic: 'Minimum',
1046
+ threshold: 1,
1047
+ treatMissingData: 'notBreaching',
1048
+ },
1049
+ {
1050
+ ...primary,
1051
+ dimensionNames: ['Slice', 'Stage'],
1052
+ dimensionValues: { Slice: name, Stage: '{stage}' },
1053
+ evaluationPeriods: 1,
1054
+ id: 'sql-outbox-attempts',
1055
+ metricName: 'SqlOutboxMaxAttempts',
1056
+ name: `${name}-{stage}-sql-outbox-attempts`,
1057
+ namespace: 'EBK/Async',
1058
+ period: 3600,
1059
+ runbook: 'docs/runbooks/async-recovery.md',
1060
+ signal: 'metric',
1061
+ statistic: 'Maximum',
1062
+ threshold: 10,
1063
+ treatMissingData: 'notBreaching',
1064
+ },
1065
+ ]
1066
+ : []),
858
1067
  {
859
1068
  ...primary,
860
1069
  dimensionNames: ['Component', 'Feature', 'Operation', 'Slice', 'Stage'],
@@ -898,9 +1107,9 @@ function deployedAlarmResources(name) {
898
1107
  ];
899
1108
  }
900
1109
  /** @internal */
901
- function renderProductReliabilityFiles(name, pascalName, reviewedAt) {
1110
+ function renderProductReliabilityFiles(name, pascalName, reviewedAt, sql = false) {
902
1111
  const contract = reliabilityContract(name, reviewedAt);
903
- const alarms = alarmRegistrations(name);
1112
+ const alarms = alarmRegistrations(name, sql);
904
1113
  const exercises = reliabilityExercises(name);
905
1114
  return {
906
1115
  'docs/reliability/slo.json': (0, built_in_plan_shared_js_1.json)(contract),
@@ -917,7 +1126,7 @@ CloudWatch resources do not rewrite it.
917
1126
  'operations/reliability.json': (0, built_in_plan_shared_js_1.json)({
918
1127
  alarmRegistrations: alarms,
919
1128
  consoleIndependent: true,
920
- deployedAlarmResources: deployedAlarmResources(name),
1129
+ deployedAlarmResources: deployedAlarmResources(name, sql),
921
1130
  diagnostic: {
922
1131
  metadataSources: [
923
1132
  'alarm-state',
@@ -1314,7 +1523,8 @@ ${` evaluationPeriods: 1,
1314
1523
  {
1315
1524
  alarmActions: [input.alertTopicArn],
1316
1525
  alarmDescription: alarmDescription({
1317
- businessSymptom: 'A generated Function has a terminal invocation error.',
1526
+ businessSymptom:
1527
+ 'The generated confidential-export cleanup Function has a terminal invocation error.',
1318
1528
  id: 'function-failure',
1319
1529
  owner: '${name}-product',
1320
1530
  route: '${name}-on-call',
@@ -99,7 +99,7 @@ export function createSqlEventOutbox(dependencies: Dependencies) {
99
99
  }
100
100
  async function reconcile(context: Context) {
101
101
  const scope = dependencies.scope(); validateScope(scope);
102
- if (!scope.enabled) return { outcome: 'paused', published: 0, failed: 0, remaining: null };
102
+ if (!scope.enabled) return { outcome: 'paused', published: 0, failed: 0, remaining: null, maxAttempts: null };
103
103
  const result = await dependencies.withDatabase(context, 'async-outbox-relay', async (client) => {
104
104
  await client.query('BEGIN');
105
105
  try {
@@ -124,9 +124,13 @@ export function createSqlEventOutbox(dependencies: Dependencies) {
124
124
  published_at = now(), expires_at = now() + interval '30 days' WHERE idempotency_key = $1 AND published_at IS NULL\`, [row.id]);
125
125
  published++;
126
126
  }
127
- const count = await client.query<{ count: string }>('SELECT count(*)::text AS count FROM data_lifecycle_outbox WHERE published_at IS NULL');
127
+ const count = await client.query<{ count: string; max_attempts: string }>(\`SELECT count(*)::text AS count,
128
+ coalesce(max(attempts), 0)::text AS max_attempts FROM data_lifecycle_outbox WHERE published_at IS NULL\`);
128
129
  const remaining = Number(count.rows[0]?.count);
129
130
  if (!Number.isSafeInteger(remaining) || remaining < 0) throw new Error('SQL outbox backlog is invalid.');
131
+ // Reporting attempts is telemetry only; it never expires or drops an intent.
132
+ const attempts = Number(count.rows[0]?.max_attempts);
133
+ const maxAttempts = Number.isSafeInteger(attempts) && attempts >= 0 ? attempts : null;
130
134
  // Bounded control-metadata cleanup; pending intents never expire. Use
131
135
  // row locks so overlapping relays cannot delete another worker's batch.
132
136
  const pruned = await client.query(\`DELETE FROM data_lifecycle_outbox WHERE idempotency_key IN (
@@ -135,7 +139,7 @@ export function createSqlEventOutbox(dependencies: Dependencies) {
135
139
  ORDER BY expires_at, idempotency_key LIMIT 100 FOR UPDATE SKIP LOCKED
136
140
  )\`);
137
141
  await client.query('COMMIT');
138
- return { outcome: failed ? 'retry-required' : remaining ? 'pending' : 'converged', published, failed, remaining, pruned: pruned.rowCount ?? 0 };
142
+ return { outcome: failed ? 'retry-required' : remaining ? 'pending' : 'converged', published, failed, remaining, maxAttempts, pruned: pruned.rowCount ?? 0 };
139
143
  } catch (error) { await client.query('ROLLBACK').catch(() => undefined); throw error; }
140
144
  });
141
145
  // Continue a healthy backlog promptly; failures remain retryable and recover hourly.
@@ -162,12 +166,21 @@ export const sqlEventOutbox = createSqlEventOutbox({
162
166
  });
163
167
  `,
164
168
  'async/sql-outbox-relay.ts': `import { randomUUID } from 'node:crypto';
169
+ import { createMetricEmitter } from '@empire-builder-kit/runtime/observability';
165
170
  import { sqlEventOutbox } from '../database/sql-events.js';
166
171
 
172
+ const metrics = createMetricEmitter();
173
+
167
174
  export async function handler(event: { id?: string } = {}) {
168
175
  const started = Date.now();
169
176
  const result = await sqlEventOutbox.reconcile({ correlationId: event.id ?? randomUUID() });
170
177
  console.info(JSON.stringify({ kind: 'ebk-sql-outbox-relay', ...result, durationMilliseconds: Date.now() - started }));
178
+ // Observational gauges precede the failure signal. A paused relay reads
179
+ // nothing and reports no backlog; a lost metric never changes delivery.
180
+ for (const [metricName, value] of [['SqlOutboxBacklog', result.remaining], ['SqlOutboxMaxAttempts', result.maxAttempts]] as const) {
181
+ if (value !== null) metrics.emit({ namespace: 'EBK/Async', metricName, unit: 'Count', value,
182
+ dimensions: { slice: '${name}', stage: process.env.EBK_STAGE ?? 'local' } });
183
+ }
171
184
  if (result.failed) throw new Error('SQL outbox retains failed deliveries; inspect content-free diagnostics and retry.');
172
185
  return result;
173
186
  }
@@ -240,7 +253,13 @@ Use stable event IDs and caller idempotency for repeat requests.
240
253
  Only after commit does a content-free notification wake the SQL relay. A missed
241
254
  notification does not undo committed data: the result reports reconciliation-
242
255
  required, and the retained outbox is recovered on the configured schedule.
243
- Pending intents are not discarded on age alone. Published intents retain thirty
256
+ Pending intents are not discarded on age alone. A relay run that retains a
257
+ failed delivery fails its invocation (\`sql-outbox-relay-errors\`). Each enabled
258
+ run reports the global unpublished count as \`SqlOutboxBacklog\` and the highest
259
+ unpublished attempt count as \`SqlOutboxMaxAttempts\`; \`sql-outbox-backlog\`
260
+ alarms when every run in two consecutive hours ended with a backlog and
261
+ \`sql-outbox-attempts\` at ten attempts. These are observations, never delivery
262
+ rules. See docs/runbooks/async-recovery.md. Published intents retain thirty
244
263
  days of replay metadata. Each enabled SQL relay invocation prunes at most one
245
264
  hundred expired published intents in the same transaction. Pausing SQL recovery
246
265
  also defers this cleanup. Pending intents are never pruned. Delivery is at least
@@ -149,7 +149,7 @@ function renderSliceFiles(name, scope, runtimeVersion, ownership, dataPolicyRevi
149
149
  ['API', (0, built_in_plan_api_files_js_1.renderProductApiFiles)(name, pascalName, scope.slice(1), sql)],
150
150
  [
151
151
  'reliability and incident response',
152
- (0, built_in_plan_reliability_files_js_1.renderProductReliabilityFiles)(name, pascalName, dataPolicyReviewedAt),
152
+ (0, built_in_plan_reliability_files_js_1.renderProductReliabilityFiles)(name, pascalName, dataPolicyReviewedAt, sql),
153
153
  ],
154
154
  [
155
155
  'product infrastructure',
@@ -351,7 +351,9 @@ const SEEDED_RECONCILIATION = {
351
351
  'alarm-registration',
352
352
  'database-capacity-dynamodb-table-and-pending-events-throttle-metrics',
353
353
  'diagnostic-authority',
354
+ 'event-delivery-alarms-and-hourly-outbox-backlog-period',
354
355
  'incident-remediation',
356
+ 'performance-load-cleanup-backlog-two-evaluation-periods',
355
357
  ],
356
358
  },
357
359
  'operations/release.json': {