@edgehero/pi-dispatch 1.10.3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/.env.example +300 -148
  2. package/README.md +50 -0
  3. package/deploy/com.pi-dispatch.worker.plist +9 -3
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +11 -0
  15. package/deploy/worker-env-wrapper.sh +60 -34
  16. package/deploy/worker.service +18 -8
  17. package/package.json +14 -4
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4701 -414
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +455 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +222 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-registry.mjs +29 -2
  54. package/src/identity.mjs +29 -4
  55. package/src/image-preflight.mjs +46 -11
  56. package/src/image-ref.mjs +21 -0
  57. package/src/index.mjs +363 -13
  58. package/src/init.mjs +197 -38
  59. package/src/job-user.mjs +252 -0
  60. package/src/json-duplicates.mjs +204 -0
  61. package/src/live-probes.mjs +1020 -0
  62. package/src/materialize.mjs +4 -11
  63. package/src/netns-keeper.mjs +264 -0
  64. package/src/on-failure.mjs +119 -0
  65. package/src/outbox.mjs +7 -0
  66. package/src/podman-stack.mjs +1304 -0
  67. package/src/prepare-github.mjs +6 -6
  68. package/src/prepare-local.mjs +51 -17
  69. package/src/prepare.mjs +27 -6
  70. package/src/processor.mjs +505 -26
  71. package/src/provider-key.mjs +41 -0
  72. package/src/provider-steering.mjs +144 -0
  73. package/src/queue.mjs +35 -8
  74. package/src/redact.mjs +84 -0
  75. package/src/reserved-env.mjs +7 -3
  76. package/src/retention-sweep.mjs +178 -0
  77. package/src/run-container.mjs +181 -14
  78. package/src/run-history.mjs +105 -16
  79. package/src/runtime-observations.mjs +1152 -0
  80. package/src/runtime-settings.mjs +13 -8
  81. package/src/sandbox-cli.mjs +100 -95
  82. package/src/sandbox-store.mjs +612 -45
  83. package/src/sandbox.mjs +1459 -37
  84. package/src/schedules.mjs +16 -3
  85. package/src/secret-profiles.mjs +2 -1
  86. package/src/secrets.mjs +23 -6
  87. package/src/service-env.mjs +247 -0
  88. package/src/service.mjs +618 -28
  89. package/src/session-store.mjs +678 -53
  90. package/src/start.mjs +1348 -326
  91. package/src/transient.mjs +240 -0
  92. package/src/triggers-file.mjs +71 -15
  93. package/src/triggers.mjs +176 -19
  94. package/src/up.mjs +1399 -85
  95. package/src/valkey-auth.mjs +529 -0
  96. package/src/valkey-endpoint.mjs +367 -0
  97. package/src/watch-closer.mjs +158 -0
@@ -46,39 +46,57 @@ fi
46
46
  # own CLI, not just from the daemon: `pi-dispatch service stop` on macOS is `launchctl kill SIGTERM` at
47
47
  # this pid.
48
48
  #
49
- # The other window is two instructions wide, and is closed by the re-send after `child=$!` below. The
50
- # handler is a FUNCTION rather than a trap string because it is installed twice -- here, and again after
51
- # the sourcing -- and one behaviour spelled out in two places is one behaviour that can drift.
52
- signaled=0
53
- child=
54
- wrapper_on_stop() {
55
- signaled=1
56
- # `child` is empty until the fork below has been assigned, and `kill -TERM ""` kills nothing and
57
- # fails silently, so a stop arriving before then has no pid to reach. It is not lost: the re-send
58
- # after `child=$!` re-delivers it, and the launch gate refuses to start at all if nothing was
59
- # started yet.
60
- [ -n "$child" ] && kill -TERM "$child" 2>/dev/null
61
- # Never leave a nonzero status behind. `rc=$?` is read immediately after the `wait` this interrupts,
62
- # and the double wait at the bottom keys on rc >= 128.
63
- return 0
49
+ # The other window is two instructions wide, and is closed by the re-send after `child=$!` below.
50
+ #
51
+ # NOTHING THIS WRAPPER READS IS LEFT WHERE ./.env CAN ASSIGN IT (issue #470). ./.env is sourced into this
52
+ # shell, so any line in it can set any variable this script uses: `env_setup=/x.sh` named a script the
53
+ # wrapper then ran, `signaled=1` made it exit 0 without starting the worker, and `child=<pid>` pointed the
54
+ # stop at another process. So the rule is: every variable below is assigned AFTER the load; the one value
55
+ # it needs from before the load (PI_ENV_SETUP, as the unit set it) travels across the load in the
56
+ # positional parameters, which no assignment line can reach; and a stop DURING the preparation is handled
57
+ # by `wrapper_stopped_early`, which reads no variable at all and simply does not start the worker.
58
+ wrapper_stopped_early() {
59
+ # A stop that arrived while the environment was being prepared is honoured by NOT STARTING. Launching
60
+ # would hand the service manager a worker it has already asked to go away: it would reserve a budget
61
+ # slot and take a job, and then need a drain nobody is waiting for. Exit 0 because 0 is the only code
62
+ # launchd's KeepAlive/SuccessfulExit=false leaves stopped -- the same reason the exit-2 conversion at the
63
+ # bottom exists. Not 2, because nothing was refused; not 1, because nothing failed; the manager's own
64
+ # instruction was carried out, and this says so rather than exiting mute. The shell runs a trap only
65
+ # after the command in progress returns, so a setup script's network call is not cut off mid-write.
66
+ echo "worker-env-wrapper: stopped before the worker started -- a stop signal arrived while the environment was being prepared, so the command was never launched; exiting 0 (nothing to restart)" >&2
67
+ exit 0
64
68
  }
65
- trap wrapper_on_stop TERM INT
69
+ trap wrapper_stopped_early TERM INT
66
70
 
67
71
  # The env-setup seam (issue #209): `pi-dispatch service render|install --env-setup <path>` puts an
68
72
  # operator-typed path here -- the plist's EnvironmentVariables dict on macOS, nssm's AppEnvironmentExtra
69
73
  # on Windows -- so a secrets manager can fill this process's environment without anyone hand-editing a
70
74
  # rendered unit. Captured BEFORE ./.env is sourced, on purpose: the path is UNIT configuration, and a
71
- # `.env` line must never be able to name a script this wrapper then runs.
72
- env_setup="${PI_ENV_SETUP:-}"
75
+ # `.env` line must never be able to name a script this wrapper then runs. It is held as "$1" across the
76
+ # load (a `PI_ENV_SETUP=` or `env_setup=` line there reaches neither), and the command is "$2" onwards
77
+ # until the `shift` below.
78
+ set -- "${PI_ENV_SETUP:-}" "$@"
73
79
 
74
80
  if [ -f ./.env ]; then
75
81
  set -a; . ./.env; set +a
76
- elif [ -z "$env_setup" ]; then
82
+ elif [ -z "$1" ]; then
77
83
  echo "worker-env-wrapper: .env not found in $PWD -- this wrapper must be started in the deployment folder (the unit's WorkingDirectory / nssm AppDirectory); it no longer guesses a location from its own path" >&2
78
84
  exit 1
79
85
  else
80
86
  # Only a configured seam earns this: the environment demonstrably comes from somewhere else.
81
- echo "worker-env-wrapper: no .env in $PWD -- the environment comes from $env_setup (PI_ENV_SETUP)" >&2
87
+ echo "worker-env-wrapper: no .env in $PWD -- the environment comes from $1 (PI_ENV_SETUP)" >&2
88
+ fi
89
+
90
+ # AFTER THE LOAD, every variable this wrapper reads is assigned, so no assignment line of ./.env
91
+ # reaches one. The worker sees the unit's PI_ENV_SETUP too, never one the file set (the setup script,
92
+ # which is unit configuration, may still set its own).
93
+ env_setup=$1
94
+ shift
95
+ if [ -n "$env_setup" ]; then
96
+ PI_ENV_SETUP=$env_setup
97
+ export PI_ENV_SETUP
98
+ else
99
+ unset PI_ENV_SETUP
82
100
  fi
83
101
 
84
102
  # AFTER ./.env, deliberately. The manager is the newer source of truth, so a stale key left in the file
@@ -102,23 +120,31 @@ if [ -n "$env_setup" ]; then
102
120
  set +a
103
121
  fi
104
122
 
105
- # RE-ASSERTED after the sourcing, and this is not belt-and-braces. A sourced script runs in THIS shell,
106
- # so a `trap ... TERM` inside one REPLACES the handler above and the drain silently disappears -- a
107
- # manager's cleanup helper does exactly that. One line restores it. What it cannot undo is a script that
108
- # IGNORES TERM (`trap '' TERM`): a signal discarded while it was ignored is already gone, and the child
109
- # forked below would inherit SIG_IGN and be unable to trap TERM at all. That is why docs/secrets.md now
110
- # tells operators not to touch signals in a setup script.
123
+ # THE FORWARDING HANDLER, installed after the sourcing, and the placement is not belt-and-braces. A
124
+ # sourced script runs in THIS shell, so a `trap ... TERM` inside one REPLACES whatever handler is up and
125
+ # the drain would silently disappear -- a manager's cleanup helper does exactly that. Installing it here
126
+ # restores it. Its two variables are assigned here too, after the load, for the reason given at the
127
+ # top. What it cannot undo is a script that IGNORES TERM (`trap '' TERM`): a signal discarded while it
128
+ # was ignored is already gone, and the child forked below would inherit SIG_IGN and be unable to trap
129
+ # TERM at all. That is why docs/secrets.md now tells operators not to touch signals in a setup script.
130
+ signaled=0
131
+ child=
132
+ wrapper_on_stop() {
133
+ signaled=1
134
+ # `child` is empty until the fork below has been assigned, and `kill -TERM ""` kills nothing and
135
+ # fails silently, so a stop arriving before then has no pid to reach. It is not lost: the re-send
136
+ # after `child=$!` re-delivers it, and the launch gate refuses to start at all if nothing was
137
+ # started yet.
138
+ [ -n "$child" ] && kill -TERM "$child" 2>/dev/null
139
+ # Never leave a nonzero status behind. `rc=$?` is read immediately after the `wait` this interrupts,
140
+ # and the double wait at the bottom keys on rc >= 128.
141
+ return 0
142
+ }
111
143
  trap wrapper_on_stop TERM INT
112
144
 
113
- # A stop that arrived while the environment was being prepared is honoured by NOT STARTING. Launching now
114
- # would hand the service manager a worker it has already asked to go away: it would reserve a budget slot
115
- # and take a job, and then need a drain nobody is waiting for. Exit 0 because 0 is the only code launchd's
116
- # KeepAlive/SuccessfulExit=false leaves stopped -- the same reason the exit-2 conversion at the bottom
117
- # exists. Not 2, because nothing was refused; not 1, because nothing failed; the manager's own instruction
118
- # was carried out, and this says so rather than exiting mute.
145
+ # A stop between the line above and the fork: the same answer as one during the preparation.
119
146
  if [ "$signaled" -eq 1 ]; then
120
- echo "worker-env-wrapper: stopped before the worker started -- a stop signal arrived while the environment was being prepared, so the command was never launched; exiting 0 (nothing to restart)" >&2
121
- exit 0
147
+ wrapper_stopped_early
122
148
  fi
123
149
 
124
150
  # `exec` is deliberately GONE here (it used to hand this shell's pid straight to node): intercepting
@@ -13,20 +13,28 @@
13
13
  # WorkingDirectory / EnvironmentFile / User / node path below are PLACEHOLDERS: set them to wherever
14
14
  # you cloned the repo and whoever owns it.
15
15
  #
16
- # PI_LOGS_DIR (run-history records; default OS-temp /pi-dispatch/logs) is created and written by the
17
- # worker at boot, so it must be writable by User= (pi) -- do NOT pre-create it as root, or the non-root
18
- # worker hits EACCES at boot. Set via `.env` (EnvironmentFile), no unit change needed; the default path
19
- # avoids colliding with the daemon's own logs/worker.out.log.
16
+ # PI_LOGS_DIR (run-history records) and PI_SETTINGS_FILE (the runtime-tunable settings overlay) both
17
+ # default to ~/.pi-dispatch, which is DURABLE across a reboot. Both are created and written by the
18
+ # worker at boot, so they must be writable by User= (pi) -- do NOT pre-create them as root, or the
19
+ # non-root worker hits EACCES at boot.
20
20
  #
21
- # PI_SETTINGS_FILE is the runtime-tunable settings overlay (default under OS temp, which may be wiped on
22
- # reboot) -- point it at a durable path in production. Set via `.env` (EnvironmentFile), no unit change
23
- # needed; it is worker-owned and never belongs in the container env allowlist.
21
+ # SET BOTH EXPLICITLY HERE, and this is the one setting on this page you should not skip. The default
22
+ # is per USER, and User= below is very likely not the account you run `/dispatch` from: leave them
23
+ # unset and the worker writes ~pi/.pi-dispatch while your panel reads ~you/.pi-dispatch, so the run
24
+ # list shows nothing and every cap you set from the panel lands in a file this worker never opens.
25
+ # `pi-dispatch up` writes both into `.env` for you, as the ACCOUNT DEFAULT resolved by whoever ran it.
26
+ # If your own shell sets either to a relative path or one inside the deployment folder, `up` refuses to
27
+ # persist it and says so, because the retention sweep below is why. If the service runs as
28
+ # another account than the one that ran `up`, check that account can write there. Set via `.env`
29
+ # (EnvironmentFile), no unit change needed; both are worker-owned and never belong in the container
30
+ # env allowlist. Do not point PI_LOGS_DIR at the daemon's own logs/ directory: the retention sweep
31
+ # deletes every .log and .json past its window and would eat worker.out.log.
24
32
 
25
33
  [Unit]
26
34
  Description=pi-dispatch worker (drains the job queue on the host; launches job containers via docker)
27
35
  After=network-online.target docker.service
28
36
  Wants=network-online.target
29
- # Valkey must be reachable (docker compose -f deploy/docker-compose.yml up -d), but it is a separate
37
+ # Valkey must be reachable (on this host, `pi-dispatch up` in the deployment folder starts it), but it is a separate
30
38
  # unit/container -- not ordered here since it may be remote.
31
39
  # Crash-loop bound: at most StartLimitBurst restarts within StartLimitIntervalSec, then systemd stops
32
40
  # trying. A restart loop against a paid provider is a bill, not just log noise. Pairs with
@@ -36,6 +44,8 @@ StartLimitBurst=5
36
44
 
37
45
  [Service]
38
46
  Type=simple
47
+ # On native Linux a job container runs as this account's uid (issue #341), so a local-folder trigger's folder
48
+ # must be writable by it, and the account must not be root.
39
49
  User=pi
40
50
  WorkingDirectory=/opt/pi-dispatch
41
51
  EnvironmentFile=/opt/pi-dispatch/.env
package/package.json CHANGED
@@ -1,8 +1,8 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "1.10.3",
3
+ "version": "2.0.0",
4
4
  "type": "module",
5
- "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
5
+ "description": "The pi-dispatch worker and CLI: runs the pi coding agent as a self-hosted service, one locked down Docker or Podman container per job, with spend caps checked before anything is spent, plus init, up, doctor and service install.",
6
6
  "keywords": [
7
7
  "pi",
8
8
  "pi-coding-agent",
@@ -19,7 +19,8 @@
19
19
  "bullmq",
20
20
  "cron",
21
21
  "docker",
22
- "cli"
22
+ "cli",
23
+ "podman"
23
24
  ],
24
25
  "license": "MIT",
25
26
  "author": "Rob Boerman",
@@ -43,6 +44,11 @@
43
44
  ".": "./src/index.mjs",
44
45
  "./config": "./src/config.mjs",
45
46
  "./exit-code": "./src/exit-code.mjs",
47
+ "./entry": "./src/entry.mjs",
48
+ "./git-hardening": "./src/git-hardening.mjs",
49
+ "./env-file": "./src/env-file.mjs",
50
+ "./service-env": "./src/service-env.mjs",
51
+ "./podman-stack": "./src/podman-stack.mjs",
46
52
  "./flow-gate": "./src/flow-gate.mjs",
47
53
  "./materialize": "./src/materialize.mjs",
48
54
  "./open-browser": "./src/open-browser.mjs",
@@ -50,6 +56,7 @@
50
56
  "./queue": "./src/queue.mjs",
51
57
  "./capabilities": "./src/capabilities.mjs",
52
58
  "./connection": "./src/connection.mjs",
59
+ "./valkey-auth": "./src/valkey-auth.mjs",
53
60
  "./job-id": "./src/job-id.mjs",
54
61
  "./forges": "./src/forges.mjs",
55
62
  "./backend-local": "./src/backend-local.mjs",
@@ -57,6 +64,7 @@
57
64
  "./backend-registry": "./src/backend-registry.mjs",
58
65
  "./container-spec": "./src/container-spec.mjs",
59
66
  "./backend-conformance": "./src/backend-conformance.mjs",
67
+ "./live-probes": "./src/live-probes.mjs",
60
68
  "./triggers": "./src/triggers.mjs",
61
69
  "./triggers-file": "./src/triggers-file.mjs",
62
70
  "./packages": "./src/packages.mjs",
@@ -64,6 +72,7 @@
64
72
  "./scoped-limits": "./src/scoped-limits.mjs",
65
73
  "./wait-for": "./src/wait-for.mjs",
66
74
  "./wait-state": "./src/wait-state.mjs",
75
+ "./cancel-state": "./src/cancel-state.mjs",
67
76
  "./host-registry": "./src/host-registry.mjs",
68
77
  "./identity": "./src/identity.mjs",
69
78
  "./gitlab-identity": "./src/gitlab-identity.mjs",
@@ -78,7 +87,8 @@
78
87
  "./subscriptions": "./src/subscriptions.mjs",
79
88
  "./budget": "./src/budget.mjs",
80
89
  "./pricing": "./src/pricing.mjs",
81
- "./scheduler-stall-guard": "./src/scheduler-stall-guard.mjs"
90
+ "./scheduler-stall-guard": "./src/scheduler-stall-guard.mjs",
91
+ "./watch-closer": "./src/watch-closer.mjs"
82
92
  },
83
93
  "engines": {
84
94
  "node": ">=22.19.0"
@@ -24,6 +24,7 @@
24
24
  import { configError } from "./config.mjs";
25
25
  import { InfraRetry } from "./processor.mjs";
26
26
  import { fetchFailureReason } from "./gitlab-identity.mjs";
27
+ import { isDeterminateFetchFailure } from "./transient.mjs";
27
28
 
28
29
  const API_VERSION = "7.1";
29
30
 
@@ -42,6 +43,15 @@ export function makeAzureHost({ orgUrl, fetchFn = fetch } = {}) {
42
43
  try {
43
44
  res = await fetchFn(url, { headers: { Authorization: authHeader(token), accept: "application/json" }, redirect: "error" });
44
45
  } catch (err) {
46
+ // The one condition where the identity modules and this one now agree, because the issue's
47
+ // acceptance asks for exactly that (#316): a TLS trust failure, a protocol mismatch and a URL
48
+ // that always redirects are the operator's, and retrying them twice before failing tells
49
+ // nobody anything. Everything else here stays unconditionally retryable, which is right for a
50
+ // per-job call behind the queue: a wrong retry costs one more attempt, a wrong refusal costs
51
+ // the delivery and posts publicly that the deployment is misconfigured.
52
+ if (isDeterminateFetchFailure(err)) {
53
+ throw configError(`azure-host: GET ${path} failed (${fetchFailureReason(err)})`);
54
+ }
45
55
  throw new InfraRetry(`azure-host: GET ${path} failed (${fetchFailureReason(err)})`);
46
56
  }
47
57
  if (!res.ok) {
@@ -157,6 +167,15 @@ export function makeAzureHost({ orgUrl, fetchFn = fetch } = {}) {
157
167
  redirect: "error",
158
168
  });
159
169
  } catch (err) {
170
+ // The one condition where the identity modules and this one now agree, because the issue's
171
+ // acceptance asks for exactly that (#316): a TLS trust failure, a protocol mismatch and a URL
172
+ // that always redirects are the operator's, and retrying them twice before failing tells
173
+ // nobody anything. Everything else here stays unconditionally retryable, which is right for a
174
+ // per-job call behind the queue: a wrong retry costs one more attempt, a wrong refusal costs
175
+ // the delivery and posts publicly that the deployment is misconfigured.
176
+ if (isDeterminateFetchFailure(err)) {
177
+ throw configError(`azure-host: POST ${path} failed (${fetchFailureReason(err)})`);
178
+ }
160
179
  throw new InfraRetry(`azure-host: POST ${path} failed (${fetchFailureReason(err)})`);
161
180
  }
162
181
  if (!res.ok) {
@@ -15,6 +15,7 @@
15
15
  */
16
16
 
17
17
  import { configError } from "./config.mjs";
18
+ import { isDeterminateFetchFailure, isJsonContentType, isTransientStatus, responseHeaderReader, transientError } from "./transient.mjs";
18
19
  import { fetchFailureReason } from "./gitlab-identity.mjs";
19
20
 
20
21
  /**
@@ -33,17 +34,32 @@ export async function resolveAzureSelfId({ orgUrl, token, fetchFn = fetch }) {
33
34
  try {
34
35
  res = await fetchFn(url, { headers: { Authorization: auth, accept: "application/json" }, redirect: "error" });
35
36
  } catch (err) {
36
- throw configError(`could not resolve the azure bot identity from ${url}: ${fetchFailureReason(err)}`);
37
+ // Transient, EXCEPT for the trust and redirect faults below. azure-host.mjs retries every
38
+ // rejection unconditionally; that is right for a job and wrong for a boot, which is the whole
39
+ // distinction issue #316 turns on.
40
+ if (isDeterminateFetchFailure(err)) throw configError(`could not resolve the azure bot identity from ${url}: ${fetchFailureReason(err)}`);
41
+ throw transientError(`could not resolve the azure bot identity from ${url}: ${fetchFailureReason(err)}`, err);
37
42
  }
38
43
  if (!res.ok) {
39
44
  // The status only. An Azure error body can echo the request, and the request carried the token.
45
+ if (isTransientStatus(res.status, responseHeaderReader(res))) {
46
+ throw transientError(`could not resolve the azure bot identity: connectionData returned ${res.status}`);
47
+ }
40
48
  throw configError(`could not resolve the azure bot identity: connectionData returned ${res.status}`);
41
49
  }
42
50
  let body;
43
51
  try {
44
52
  body = await res.json();
45
53
  } catch (err) {
46
- throw configError(`could not resolve the azure bot identity: unparseable JSON from connectionData (${err?.message ?? "unknown"})`);
54
+ // A body that will not parse is ambiguous in the expensive direction, so the CONTENT TYPE
55
+ // decides it. A truncated JSON body is transient. An HTML page is not: it is a Cloudflare
56
+ // Access or OIDC portal answering instead of the forge, and no amount of restarting gets past
57
+ // one (issue #316).
58
+ const contentType = responseHeaderReader(res);
59
+ if (contentType("content-type") !== undefined && !isJsonContentType(contentType)) {
60
+ throw configError(`could not resolve the azure bot identity: connectionData answered ${String(contentType("content-type"))} instead of JSON. Azure answers an expired or invalid PAT with 203 and a sign-in page, which is what this usually is`);
61
+ }
62
+ throw transientError(`could not resolve the azure bot identity: unparseable JSON from connectionData (${err?.message ?? "unknown"})`, err);
47
63
  }
48
64
 
49
65
  const user = body?.authenticatedUser ?? {};
@@ -14,7 +14,9 @@
14
14
  *
15
15
  * WHAT IT CAN AND CANNOT DO, stated at its true size. It verifies the SHAPE of a bundle, the INTERNAL
16
16
  * CONSISTENCY of its declaration, and the behaviour of whatever the caller's PROBES produce. It cannot
17
- * start a real container, reach a real daemon, or prove a kernel enforced anything.
17
+ * start a real container, reach a real daemon, or prove a kernel enforced anything itself: the eight properties a
18
+ * container read can reach arrive through a `readBack` probe (READ_BACK_BY_A_LIVE_PROBE), which for `local` is
19
+ * `doctor --live`.
18
20
  *
19
21
  * THE PROBES ARE THE ADAPTER'S OWN CODE, and that is a real limit rather than a detail. How you make a
20
22
  * container exit 2, or make an enumeration fail, is the runtime's business and cannot be written
@@ -199,7 +201,9 @@ function checkTransfers(backend) {
199
201
  * `probe` is the one thing an adapter must supply: `(backend, { exitCode, aborted }) => result`, arranging
200
202
  * for the backend's own `runContainer` to produce a container that exits that way. It cannot be written
201
203
  * generically -- how you make a container exit 2 is the runtime's business -- and it is the reason this is a
202
- * function taking probes rather than a fixed suite.
204
+ * function taking probes rather than a fixed suite. `withBrokenEnumeration` drives the reaper's failure path, and
205
+ * `readBack(backend)` reports the eight properties read off a live container (see READ_BACK_BY_A_LIVE_PROBE); each
206
+ * one missing abstains.
203
207
  *
204
208
  * Returns `{ ok, findings }`. `ok` is false if ANY check failed; a check that abstained does not fail the
205
209
  * run but is reported, because a property nobody could verify is not a property that was verified.
@@ -236,27 +240,76 @@ async function conformance(backend, probes) {
236
240
  findings.push(abstain("abortable", "no `probe` supplied, so an abort could not be told from an OOM"));
237
241
  }
238
242
  findings.push(...(await checkReaperTriState(backend, probes)));
243
+ findings.push(...(await checkReadBack(backend, probes)));
239
244
  return { ok: findings.every((f) => f.ok), findings };
240
245
  }
241
246
 
242
247
  /**
243
- * The properties this harness DOES NOT verify, and why -- printed alongside the findings so a green run is
244
- * never mistaken for a conformant backend.
248
+ * The properties a LIVE PROBE reads back off a real container (issue #278), rather than this harness checking a
249
+ * report or the declaration itself. For `local` that probe ships: `doctor --live` (`live-probes.mjs`). For any other
250
+ * runtime the adapter supplies `readBack(backend)`, which must return the same verdicts -- `{ [property]: { ok,
251
+ * warn?, detail } }`, or `live-probes.mjs`'s array of `{ property, ok, warn?, detail }` -- read off a container its
252
+ * own `runContainer` path started. Like `probe`, a readBack that fabricates its answer passes while proving nothing.
253
+ */
254
+ export const READ_BACK_BY_A_LIVE_PROBE = Object.freeze(["isolation", "ephemeral", "mountSet", "egress", "jobToJobIsolation", "imagePinning", "nonRoot", "localFolders"]);
255
+
256
+ /**
257
+ * THE READ-BACK. Abstains without a report, and per property for anything the report does not cover or could not
258
+ * read, because a property nobody read is not one that held. A property read back as NOT holding fails when the
259
+ * backend declares it `enforced` or `asserted` -- both claim it is provided -- and passes against `absent`, which
260
+ * claimed nothing.
261
+ */
262
+ async function checkReadBack(backend, { readBack }) {
263
+ if (typeof readBack !== "function") {
264
+ return READ_BACK_BY_A_LIVE_PROBE.map((property) => abstain(property, "no `readBack` probe supplied, so it was not read off a live container (doctor --live reads local's)"));
265
+ }
266
+ const raw = await readBack(backend);
267
+ // OWN properties only, and a property named twice in an array is ambiguous rather than last-wins: a report that
268
+ // says a property both fails and holds must not be read as holding.
269
+ const report = Object.create(null);
270
+ const repeated = new Set();
271
+ if (Array.isArray(raw)) {
272
+ for (const v of raw) {
273
+ const property = v?.property;
274
+ if (typeof property !== "string") continue;
275
+ if (Object.hasOwn(report, property)) repeated.add(property);
276
+ report[property] = v;
277
+ }
278
+ } else if (raw && typeof raw === "object") {
279
+ for (const property of Object.keys(raw)) report[property] = raw[property];
280
+ }
281
+ const declares = backend?.declares ?? {};
282
+ const claimed = (property) => declares[property] === "enforced" || declares[property] === "asserted";
283
+ // A reading counts only through its OWN `ok`; any truthy `warn`, own or not, means not read back.
284
+ const failing = (v) => v && typeof v === "object" && Object.hasOwn(v, "ok") && v.ok === false && !v.warn;
285
+ return READ_BACK_BY_A_LIVE_PROBE.map((property) => {
286
+ if (repeated.has(property)) {
287
+ // Ambiguous, but not in the backend's favour: any failing reading among the repeats fails a claimed property.
288
+ const failed = raw.filter((v) => v?.property === property && failing(v));
289
+ if (failed.length > 0 && claimed(property)) return fail(property, `declared ${declares[property]}, and among the readings the report names more than once, one does not hold: ${failed[0].detail ?? "no detail"}`);
290
+ return abstain(property, "the readBack report names it more than once, so which reading holds is not known");
291
+ }
292
+ const got = Object.hasOwn(report, property) ? report[property] : undefined;
293
+ if (!got || typeof got !== "object" || !Object.hasOwn(got, "ok") || typeof got.ok !== "boolean") return abstain(property, "the readBack report does not cover it");
294
+ // `warn` first: a reading marked not-read-back is never a pass, whatever its `ok` says.
295
+ if (got.warn) return abstain(property, `${got.detail ?? "not read back"}`);
296
+ if (got.ok === true) return pass(property, `read back off a live container: ${got.detail ?? "holds"}`);
297
+ if (claimed(property)) {
298
+ return fail(property, `declared ${declares[property]}, and read back off a live container as not holding: ${got.detail ?? "no detail"}`);
299
+ }
300
+ return pass(property, `read back as not holding, which ${JSON.stringify(declares[property])} does not claim`);
301
+ });
302
+ }
303
+
304
+ /**
305
+ * The properties NEITHER this harness NOR a live read-back verifies, and why -- printed alongside the findings so a
306
+ * green run is never mistaken for a conformant backend.
245
307
  *
246
- * Every one of these needs a live container on the target runtime, which is exactly what this repo's own
247
- * offline CI cannot have. Naming them is the difference between a suite that is honest about its reach and
248
- * one that lets a green tick stand for something it never checked. `verify-image.sh` is the shape the
249
- * missing half would take, and it "runs ON THE HOST THAT HOLDS THE IMAGE, which is the only place it can".
308
+ * Eight of the ten that once stood here are now READ_BACK_BY_A_LIVE_PROBE (six in issue #278, `ephemeral` and
309
+ * `jobToJobIsolation` in issue #344). The two left are not container properties at all, so no container read can
310
+ * reach them.
250
311
  */
251
312
  export const UNVERIFIED_BY_THIS_HARNESS = Object.freeze({
252
- isolation: "needs a live container: read the capability set and no-new-privileges back from inside it",
253
- ephemeral: "needs two runs of the same job id and a check that no container survived the first",
254
- mountSet: "needs a live container: enumerate its mounts and assert nothing beyond the declared set",
255
- egress: "needs a live container and a blocked destination",
256
- jobToJobIsolation: "needs two live containers and an attempted connection between them",
257
- imagePinning: "needs a run against an image absent from the target runtime",
258
- nonRoot: "needs `id -u` inside a live container (`verify-image.sh` is the shape)",
259
- secretsCustody: "needs a canary secret and an audit of every log, record and forge comment the run produced",
260
- credentialTransit: "needs to observe what actually crossed the network to the runtime",
261
- localFolders: "needs a bind-mounted host folder and a write read back outside the container",
313
+ secretsCustody: "not a container property: needs a canary secret and an audit of every log, record and forge comment the run produced",
314
+ credentialTransit: "not a container property: what crossed the network to the runtime; local's is observed per job from the docker CLI's endpoint instead",
262
315
  });