@edgehero/pi-dispatch 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/.env.example +160 -0
  2. package/deploy/com.pi-dispatch.worker.plist +66 -0
  3. package/deploy/nssm-install.cmd +59 -0
  4. package/deploy/receiver.service +36 -0
  5. package/deploy/worker-env-wrapper.cmd +50 -0
  6. package/deploy/worker-env-wrapper.sh +63 -0
  7. package/deploy/worker.service +55 -0
  8. package/package.json +83 -0
  9. package/src/azure-auth.mjs +61 -0
  10. package/src/azure-host.mjs +236 -0
  11. package/src/azure-identity.mjs +63 -0
  12. package/src/azure-prompt.mjs +118 -0
  13. package/src/branch.mjs +80 -0
  14. package/src/budget.mjs +179 -0
  15. package/src/cli.mjs +208 -0
  16. package/src/config.mjs +329 -0
  17. package/src/connection.mjs +40 -0
  18. package/src/cron.mjs +94 -0
  19. package/src/docker-run.mjs +119 -0
  20. package/src/doctor.mjs +1127 -0
  21. package/src/env-allowlist.mjs +198 -0
  22. package/src/env-file.mjs +153 -0
  23. package/src/exit-code.mjs +32 -0
  24. package/src/flow-gate.mjs +82 -0
  25. package/src/forgejo-auth.mjs +77 -0
  26. package/src/forgejo-host.mjs +172 -0
  27. package/src/forgejo-identity.mjs +74 -0
  28. package/src/forgejo-prompt.mjs +123 -0
  29. package/src/forges.mjs +148 -0
  30. package/src/get-token.mjs +226 -0
  31. package/src/git-dirty.mjs +16 -0
  32. package/src/github-app-setup.mjs +517 -0
  33. package/src/github-host.mjs +159 -0
  34. package/src/github-prompt.mjs +286 -0
  35. package/src/gitlab-auth.mjs +72 -0
  36. package/src/gitlab-host.mjs +200 -0
  37. package/src/gitlab-identity.mjs +61 -0
  38. package/src/gitlab-prompt.mjs +123 -0
  39. package/src/identity.mjs +57 -0
  40. package/src/image-preflight.mjs +180 -0
  41. package/src/import-pi.mjs +451 -0
  42. package/src/index.mjs +177 -0
  43. package/src/init.mjs +77 -0
  44. package/src/job-id.mjs +100 -0
  45. package/src/materialize.mjs +138 -0
  46. package/src/outbox.mjs +179 -0
  47. package/src/packages.mjs +188 -0
  48. package/src/pause-windows.mjs +218 -0
  49. package/src/prepare-github.mjs +260 -0
  50. package/src/prepare-local.mjs +76 -0
  51. package/src/prepare.mjs +199 -0
  52. package/src/pricing.mjs +168 -0
  53. package/src/processor.mjs +360 -0
  54. package/src/queue.mjs +152 -0
  55. package/src/run-container.mjs +133 -0
  56. package/src/run-history.mjs +534 -0
  57. package/src/runtime-settings.mjs +188 -0
  58. package/src/sandbox-cli.mjs +156 -0
  59. package/src/sandbox-store.mjs +269 -0
  60. package/src/sandbox.mjs +171 -0
  61. package/src/scheduler-stall-guard.mjs +67 -0
  62. package/src/schedules.mjs +62 -0
  63. package/src/service.mjs +677 -0
  64. package/src/session-key.mjs +108 -0
  65. package/src/session-store.mjs +249 -0
  66. package/src/start.mjs +502 -0
  67. package/src/subscriptions.mjs +208 -0
  68. package/src/triggers.mjs +491 -0
  69. package/src/up.mjs +315 -0
package/.env.example ADDED
@@ -0,0 +1,160 @@
1
+ # Copy to .env and fill in. Never commit a real .env (it is gitignored).
2
+
3
+ # --- Provider credential ---
4
+ # pi supports ~30 providers; set the key for the one you use, under the variable name pi expects.
5
+ # The worker forwards ONLY the configured provider's key into the job container -- nothing else.
6
+ # Anthropic: ANTHROPIC_API_KEY (or ANTHROPIC_OAUTH_TOKEN, which takes precedence)
7
+ # OpenAI: OPENAI_API_KEY Google: GEMINI_API_KEY Groq: GROQ_API_KEY ... etc.
8
+ # You can LEAVE THIS BLANK if you are already logged into pi: when the env has no key, the worker reads the
9
+ # API key from ~/.pi/agent/auth.json (host-side) and env-injects it -- on by default, nothing to set.
10
+ # API-key logins only; an OAuth/subscription login is refused (it expires; use an API key for a service).
11
+ ANTHROPIC_API_KEY=
12
+ # PI_AUTH_FROM_PI=0 # uncomment to force env-only (fail loudly on a missing env key instead of using your pi login)
13
+
14
+ # --- Which provider/model to run by default (override per job with --provider / --model) ---
15
+ PI_PROVIDER=anthropic
16
+ # A DATED model id is deterministic; a floating alias (e.g. claude-sonnet-4-5) can change cost.
17
+ PI_MODEL=claude-sonnet-4-5-20250929
18
+
19
+ # --- Spend + concurrency guards (money bounds; all have conservative defaults) ---
20
+ PI_MAX_TURNS=30 # per-job turn cap -- pi has none of its own, so the harness imposes one
21
+ PI_DAILY_CAP=25 # max job containers started per day (mandatory window)
22
+ # PI_WEEKLY_CAP=100 # optional weekly ceiling on container starts; unset = weekly window disabled
23
+ # PI_MONTHLY_CAP=400 # optional monthly ceiling on container starts; unset = monthly window disabled
24
+ # PI_SOFT_HOLD_PCT=80 # optional soft-hold band (1-99): once any window hits this % of its cap, new starts
25
+ # pause (in-flight jobs finish) and the panel meter turns amber; unset = disabled
26
+ # PI_MAX_TOKENS= # optional per-job token budget: the runner aborts the agent once cumulative usage exceeds it.
27
+ # LAGGING (spends before it can see the total) -- a single-job runaway backstop, not before-the-spend;
28
+ # PI_MAX_TURNS stays the proactive lever. Unset = per-job token budget disabled (usage is still recorded).
29
+ # PI_DAILY_TOKEN_CAP= # optional daily token cap: once a day's recorded spend reaches it, the NEXT job is refused pre-container.
30
+ # Check-AFTER by nature (token cost is only known post-run), unlike PI_DAILY_CAP; unset = disabled
31
+ PI_CONCURRENCY=3 # how many jobs run in parallel
32
+
33
+ # --- Infrastructure ---
34
+ VALKEY_URL=redis://127.0.0.1:6379
35
+ PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may name its own with "image" in triggers.json (docs/job-image.md)
36
+ # Jobs run with --pull=never: pull or BUILD every image you name -- the worker never fetches one at job time, and doctor checks presence
37
+ # docker pull ghcr.io/edgehero/pi-job:latest && docker tag ghcr.io/edgehero/pi-job:latest pi-job:latest (or build image/Dockerfile)
38
+ # PI_JOBS_DIR= # where per-job /job inputs live (default: your OS temp dir)
39
+ # PI_LOGS_DIR= # where per-job status records (and optional raw logs) land (default: OS temp /pi-dispatch/logs)
40
+ # PI_CAPTURE_JOB_LOGS= # default 0; set 1 to ALSO write raw container output to logs/<jobId>.log -- PII-bearing (issue/comment text), host-only (never mounted into the container), off by default (opt-in)
41
+ # PI_LOG_RETENTION_DAYS= # default 30; boot-time prune of logs older than N days; 0 = keep forever
42
+ # PI_SANDBOX_DIR= # where a finished run's directory is kept so you can re-open it (default: <PI_JOBS_DIR>/sandboxes, mode 0700)
43
+ # `pi-dispatch sandbox <jobId>` starts a fresh container from the same image with the same workspace and NO credentials (docs/sandbox.md)
44
+ # A retained directory holds the run's clone plus its prompt.md/event.json -- so issue text. Host-only; never mounted into a job
45
+ # PI_SANDBOX_RETENTION_HOURS= # default 24. NOTE: 0 means OFF here -- nothing is retained and cleanup deletes as it always did
46
+ # This is the OPPOSITE of PI_LOG_RETENTION_DAYS/PI_SESSIONS_TTL_DAYS, where 0 means keep forever
47
+ # There is deliberately no keep-forever value: one repo clone per run with no ceiling is a disk bomb. Use --pin for the one run worth keeping
48
+ # PI_SANDBOX_PIN_DAYS= # default 7; `pi-dispatch sandbox <jobId> --pin` extends THAT run to now + this many days. Still swept afterwards
49
+ # PI_SANDBOX_IDLE_MINUTES= # default 30; bash's own TMOUT inside a sandbox, so a forgotten shell closes itself; 0 = no idle logout
50
+ # Honest gap: TMOUT does not tick while a foreground command runs, so a sandbox left serving an app stays up (`pi-dispatch sandbox --list` finds it)
51
+ # PI_SESSIONS_DIR= # NO DEFAULT, on purpose. Where persisted agent transcripts live, so a trigger with "resume": true can continue the session that opened the PR (docs/sessions.md)
52
+ # Unset = the feature is unavailable and an armed trigger refuses PRE-SPEND rather than running unpersisted and looking like it worked
53
+ # A transcript is the most PII-bearing thing this system stores -- issue text, file contents, tool output, the agent's own reasoning. Mode 0700, host-only, OUTSIDE any git repo
54
+ # Deliberately not defaulted into the OS temp dir the way PI_LOGS_DIR is: that is mode 1777 on POSIX
55
+ # PI_SESSIONS_TTL_DAYS= # default 14; a transcript older than this is not resumed AND is swept at boot; 0 = keep forever
56
+ # Enforced at OPEN as well as at boot: a stale transcript is a live input to a future job, not debris
57
+ # PI_SESSION_MAX_BYTES= # default 8388608 (8 MiB); a transcript larger than this is not resumed; 0 = no cap
58
+ # Not disk hygiene -- an oversized transcript is a prefill nobody sized PI_MAX_TOKENS for
59
+ # PI_SESSIONS_ALLOW_GH_SOURCE= # unset = a run.resume job REFUSES to mint under GITHUB_AUTH_SOURCE=gh, pre-spend
60
+ # That source is your whole gh login: full-scope and non-expiring, and a transcript is a FILE -- any command that echoed an auth header persists it
61
+ # Prefer GITHUB_AUTH_SOURCE=app or a short-expiry fine-grained PAT. Set exactly 1 to accept the trade explicitly (SECURITY.md, docs/sessions.md)
62
+ # Not disk hygiene -- an oversized transcript is a prefill nobody sized PI_MAX_TOKENS for
63
+ # PI_TRIGGERS_FILE= # ABSOLUTE path to the unified triggers.json, read by BOTH worker and receiver (a relative path resolves against the service's WorkingDirectory).
64
+ # Unset = cron disabled for the worker; the receiver falls back to ./triggers.json in the folder it starts from (what `pi-dispatch init` scaffolds)
65
+ # and refuses to start when neither exists (it holds the label/comment/pull_request trigger config)
66
+ # PI_PAUSE_WINDOWS_FILE= # ABSOLUTE path to pause-windows.json — "quiet hours" per folder/repo (pause runs between certain times/days/dates, auto-resume). Unset = feature off. See docs/pause-windows.md
67
+ # PI_SUBSCRIPTIONS_FILE= # path to subscriptions.json — operator-declared subscription plan prices (the admin defaults to ./subscriptions.json in its working directory). Read by the ADMIN EXTENSION only, never at job time.
68
+ # Subscription-backed providers bill 0 per run (their rate tables are all zeros), so this file is where the real price lives — cost analytics only; it changes no routing, auth, or job behavior
69
+ # PI_SETTINGS_FILE= # ABSOLUTE path to the runtime settings overlay (default: OS temp /pi-dispatch/settings.json); edited by the admin extension, read by the worker per job
70
+
71
+ # --- Reuse your existing pi setup in every job (see docs/global-pi-overlay.md) ---
72
+ # PI_GLOBAL_PI_DIR= # dir with your host pi setup (models.json/skills/APPEND_SYSTEM.md), mounted /opt/pi-global:ro into every job, layered UNDER each repo's .pi/. Unset = off. Stage it with: pi-dispatch import-pi
73
+ # PI_GLOBAL_ALLOW_EXTENSIONS= # the overlay's extensions LOAD by default (staging them with import-pi, which prints each one, is the vetting step). Set exactly 0 to keep them staged but dormant.
74
+ # Unset, empty and the legacy 1 all mean LOAD. ANY other value refuses to boot -- a typo must never silently leave code running against adversarial input with open egress.
75
+ # This knob covers the OVERLAY only. A serviced repo's own /workspace/.pi/extensions load regardless (they are default-branch, merge-gated) -- see SECURITY.md.
76
+ # PI_PACKAGES_FILE= # path to pi-packages.json (default: ./pi-packages.json; --packages-file wins). Read ONLY by `pi-dispatch import-pi --with-packages`, never at job time.
77
+ # Staged packages/ rides INSIDE PI_GLOBAL_PI_DIR -- no separate mount, no separate env dir -- and loads for EVERY job once staged; decline it PER TRIGGER with "packages": false in triggers.json, NOT by an env flag.
78
+ # Versions must be EXACT (no ^ ~ * or latest); staging uses --ignore-scripts, so a package needing a build step is staged INCOMPLETE and import-pi warns.
79
+ # PI_FORWARD_ENV= # comma-separated extra env var NAMES to forward into the container (e.g. a CUSTOM provider's key). Explicit allowlist, not a pass-through.
80
+ # GITHUB_TOKEN/GH_TOKEN are refused here -- the worker mints per-job tokens
81
+
82
+ PI_SCHEDULER_STALL_MAX=2 # tear down a scheduler after N consecutive stalls (money backstop)
83
+ # PI_DISPATCH_RUN_ROOTS= # OS-path-delimited allowlist (; on Windows, : elsewhere) of folders the model-callable dispatch_run may target; default empty = fail-closed (dispatch_run refuses every folder until you opt in)
84
+ # PI_DISPATCH_RUN_PER_HOUR=3 # per-hour cap on model-invoked dispatch_run enqueues; 0 disables the tool
85
+ # PI_DISPATCH_ASCII= # set 1 to render the admin extension's views with plain ASCII instead of box-drawing/sparkline-ramp glyphs (for glyph-width-hostile terminals); read at extension load
86
+ # PI_CHAIN_DEPTH_MAX=1 # max follow-up chain depth from a job's /outbox; 0 = chaining kill-switch
87
+ # PI_CHAIN_MAX_PER_JOB=2 # max request-<n>.json collected per completed job
88
+
89
+ # --- GitHub trigger (receiver + worker auth) ---
90
+ # Webhook receiver
91
+ WEBHOOK_SECRET=
92
+ RECEIVER_PORT=3000
93
+ RECEIVER_BIND=0.0.0.0
94
+ # Worker GitHub auth: source is gh | pat | app (default gh)
95
+ # gh = your full login scopes reach token-carrying jobs (doctor warns and names them); pat/app = narrower
96
+ GITHUB_AUTH_SOURCE=gh
97
+ # For GITHUB_AUTH_SOURCE=pat: a repo-scoped, short-expiry fine-grained PAT
98
+ GITHUB_PAT=
99
+ # For GITHUB_AUTH_SOURCE=app (optional; required for multi-tenant)
100
+ # `pi-dispatch setup github` fills all three in one browser click (App Manifest flow) and writes the PEM 0600
101
+ GITHUB_APP_ID=
102
+ GITHUB_APP_INSTALLATION_ID=
103
+ GITHUB_APP_PRIVATE_KEY_PATH=
104
+
105
+ # --- GitLab trigger (receiver + worker auth) ---
106
+ # Optional. Set these only to service GitLab projects; leaving GITLAB_TOKEN unset means no /gitlab
107
+ # endpoint exists at all, rather than one that answers 401. See docs/gitlab.md.
108
+ #
109
+ # A PROJECT access token with the `api` scope and the Developer role or above. `api` is the narrowest
110
+ # scope that can post a note -- GitLab offers no contents-vs-issues split -- so scope it to one project
111
+ # and rotate it (CONST-TOKEN-SCOPED-PER-JOB). A GROUP token reaches every project in the group.
112
+ GITLAB_TOKEN=
113
+ # Your instance root. Only for self-hosted GitLab.
114
+ GITLAB_URL=https://gitlab.com
115
+ # How the receiver verifies a delivery. REQUIRED once any GITLAB_* variable is set, and deliberately not
116
+ # defaulted -- the two are not equally strong, so it must be a choice somebody made:
117
+ # signature HMAC-SHA256 over the body. Needs GitLab 19.0+. Use this if you can.
118
+ # token a shared-secret compare. Works on any version, and proves nothing about the body.
119
+ GITLAB_WEBHOOK_MODE=
120
+ # The signing token (signature mode) or the secret token (token mode) from the webhook's settings.
121
+ GITLAB_WEBHOOK_SECRET=
122
+
123
+ # --- Forgejo / Gitea trigger (receiver + worker auth) --- issue #61
124
+ # Forgejo's webhook transport is byte-compatible with GitHub's (HMAC-SHA256 over the raw body,
125
+ # X-Hub-Signature-256), so there is no mode to choose here: there is one mechanism and it is the strong one.
126
+ # Point the webhook at /forgejo -- NOT at / -- because Forgejo also sends X-GitHub-* headers, so the path is
127
+ # the only thing that can tell the two apart, and a sender must never choose which gate it faces.
128
+ FORGEJO_URL=
129
+ # A REPOSITORY-scoped token ("Specific repositories") carrying only write:repository and write:issue.
130
+ # Narrower than GitLab's equivalent -- Forgejo has no all-or-nothing `api` scope. What it cannot do is
131
+ # expire: there is no App or installation token, so rotation is the whole mitigation
132
+ # (CONST-TOKEN-SCOPED-PER-JOB).
133
+ FORGEJO_TOKEN=
134
+ # The harness account's NUMERIC id. Required when the token above is repository-scoped, because such a
135
+ # token may not carry read:user and therefore cannot call GET /user. The receiver refuses to boot without an
136
+ # identity from one source or the other: the bot-loop guard compares against it, and an unresolved identity
137
+ # never matches -- so it would fail open silently and the harness's own comments would start more jobs.
138
+ FORGEJO_BOT_ID=
139
+ FORGEJO_WEBHOOK_SECRET=
140
+
141
+ # --- Azure DevOps trigger (receiver + worker auth) --- issue #43
142
+ # READ docs/azure-devops.md BEFORE ENABLING. Azure Service Hooks offer no HMAC of any kind: the credential
143
+ # proves the sender knew a secret and covers no bytes, there is no delivery-id header (the dedup key comes
144
+ # from the body), and there is no signed timestamp and so no replay window. OQ-015 records the residual.
145
+ # HTTPS is not optional here -- over plain HTTP the credential is on the wire in base64.
146
+ AZURE_ORG_URL=
147
+ # A PAT for a DEDICATED identity. Azure gives you a real expiry and cannot scope below the organization
148
+ # (vso.code_write reaches every repo in the org), so the bound comes from that identity's per-repository
149
+ # permissions in Project Settings -- not from the token's scopes. It also needs vso.graph, to resolve the
150
+ # actor's project membership before a job may be enqueued.
151
+ AZURE_TOKEN=
152
+ # REQUIRED once any AZURE_* variable is set, and deliberately not defaulted: both modes are shared-secret
153
+ # compares that cover no bytes, so which header carries the secret must be a choice somebody made.
154
+ # basic -- Authorization: Basic <base64>, the credential you set on the subscription
155
+ # header -- one custom header, named below
156
+ AZURE_WEBHOOK_MODE=
157
+ # For basic: the base64 of "user:password" exactly as the subscription sends it.
158
+ AZURE_WEBHOOK_SECRET=
159
+ # Required only when AZURE_WEBHOOK_MODE=header.
160
+ AZURE_WEBHOOK_HEADER=
@@ -0,0 +1,66 @@
1
+ <?xml version="1.0" encoding="UTF-8"?>
2
+ <!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
3
+ <!--
4
+ UNTESTED EXAMPLE: a starting point for the macOS (launchd) worker daemon, not a shipped, verified
5
+ unit. Adapt it. The Linux/systemd equivalent is deploy/worker.service.
6
+
7
+ ProgramArguments points at deploy/worker-env-wrapper.sh; that wrapper is what loads `.env`, because
8
+ launchd has no EnvironmentFile mechanism. NO secrets are inlined here: there is deliberately no
9
+ EnvironmentVariables dict, since that would commit credentials into this file. The wrapper reads
10
+ `.env` at runtime instead (note the ANTHROPIC_OAUTH_TOKEN over ANTHROPIC_API_KEY precedence trap
11
+ documented in the wrapper).
12
+
13
+ Graceful shutdown needs NO macOS-specific code: `launchctl bootout` sends SIGTERM, which the wrapper
14
+ forwards to node (a trap + kill; its former `exec` is gone, see the wrapper's own comments), and the
15
+ worker drains in-flight work on SIGTERM. ExitTimeOut 30 gives it room (at least the 5s docker-stop
16
+ grace) before launchd escalates to SIGKILL.
17
+
18
+ KeepAlive restarts on a crash (SuccessfulExit false) but NOT on a clean exit, so a deliberate stop
19
+ stays stopped. KeepAlive cannot exclude a single exit code the way systemd's
20
+ RestartPreventExitStatus=2 and nssm's `AppExit 2 Exit` do, so the wrapper converts EXIT_POLICY
21
+ (exit 2, a determinate config/budget refusal) into a clean exit 0: a policy refusal stays stopped
22
+ instead of relaunch-looping against a paid provider.
23
+
24
+ Per-host PLACEHOLDERS: replace /opt/pi-dispatch (the repo root, used in ProgramArguments and
25
+ WorkingDirectory) and /opt/pi-dispatch/logs (StandardOutPath, StandardErrorPath) with your paths.
26
+
27
+ One worker per host (DES-CONCURRENCY-3): parallelism is PI_CONCURRENCY inside the one process.
28
+ Requires the AOF-enabled Valkey from deploy/docker-compose.yml.
29
+
30
+ PI_LOGS_DIR (run-history records; default OS-temp /pi-dispatch/logs) is created and written by the
31
+ worker at boot, so it must be writable by the account the daemon runs as. Set via `.env` (the wrapper),
32
+ not a plist change; its default is distinct from the StandardOutPath worker.out.log below.
33
+ -->
34
+ <plist version="1.0">
35
+ <dict>
36
+ <key>Label</key>
37
+ <string>com.pi-dispatch.worker</string>
38
+
39
+ <key>ProgramArguments</key>
40
+ <array>
41
+ <string>/bin/sh</string>
42
+ <string>/opt/pi-dispatch/deploy/worker-env-wrapper.sh</string>
43
+ </array>
44
+
45
+ <key>WorkingDirectory</key>
46
+ <string>/opt/pi-dispatch</string>
47
+
48
+ <key>RunAtLoad</key>
49
+ <true/>
50
+
51
+ <key>KeepAlive</key>
52
+ <dict>
53
+ <key>SuccessfulExit</key>
54
+ <false/>
55
+ </dict>
56
+
57
+ <key>ExitTimeOut</key>
58
+ <integer>30</integer>
59
+
60
+ <key>StandardOutPath</key>
61
+ <string>/opt/pi-dispatch/logs/worker.out.log</string>
62
+
63
+ <key>StandardErrorPath</key>
64
+ <string>/opt/pi-dispatch/logs/worker.err.log</string>
65
+ </dict>
66
+ </plist>
@@ -0,0 +1,59 @@
1
+ @echo off
2
+ REM UNTESTED EXAMPLE -- a starting point for the Windows worker service, not a shipped, verified unit.
3
+ REM Adapt it. The Linux/systemd equivalent is deploy/worker.service; the macOS one is
4
+ REM deploy/com.pi-dispatch.worker.plist.
5
+ REM
6
+ REM Registers the pi-dispatch worker as a Windows service via nssm (the Non-Sucking Service Manager).
7
+ REM nssm is an operator-downloaded binary (https://nssm.cc) -- it is documented here, NOT vendored into
8
+ REM this repo; put nssm.exe on PATH before running this. The service's Application is the `.cmd` wrapper
9
+ REM (deploy/worker-env-wrapper.cmd), which loads `.env` at runtime -- so NO secrets are inlined here or
10
+ REM passed via AppEnvironmentExtra. `.env` is gitignored; nothing in this file is a credential.
11
+ REM
12
+ REM Why nssm over Task Scheduler: Task Scheduler stops a task with TerminateProcess (a hard kill), which
13
+ REM gives the worker no chance to drain the in-flight job -- it then relies on the queue's boot reaper to
14
+ REM recover the orphaned container. nssm's AppStopMethodConsole sends a real Ctrl-C first, which node
15
+ REM receives as SIGINT for a graceful drain. Task Scheduler is a weaker fallback, not the recommended
16
+ REM path.
17
+ REM
18
+ REM One worker per host (DES-CONCURRENCY-3): parallelism is PI_CONCURRENCY inside the one process, not
19
+ REM multiple services. Requires the AOF-enabled Valkey from deploy/docker-compose.yml.
20
+ REM
21
+ REM Per-host PLACEHOLDERS: set SERVICE / REPO / LOGDIR below for your host before running.
22
+ REM
23
+ REM PI_LOGS_DIR (run-history records; default OS-temp \pi-dispatch\logs) is created and written by the
24
+ REM worker at boot, so it must be writable by the service account. Set via `.env` (the wrapper), not a
25
+ REM change here; its default avoids colliding with the nssm LOGDIR worker.out log set below.
26
+ REM
27
+ REM PI_SETTINGS_FILE is the runtime-tunable settings overlay (default under OS temp, which may be wiped
28
+ REM on reboot) -- point it at a durable path in production. Set via `.env` (the wrapper), not a change
29
+ REM here; it is worker-owned and never belongs in the container env allowlist.
30
+
31
+ setlocal
32
+
33
+ set "SERVICE=pi-dispatch-worker"
34
+ set "REPO=C:\pi-dispatch"
35
+ set "LOGDIR=C:\pi-dispatch\logs"
36
+
37
+ REM Application is the wrapper (loads `.env`), not node directly and not an env dict with real values.
38
+ nssm install %SERVICE% "%REPO%\deploy\worker-env-wrapper.cmd"
39
+ nssm set %SERVICE% AppDirectory "%REPO%"
40
+ nssm set %SERVICE% AppStdout "%LOGDIR%\worker.out.log"
41
+ nssm set %SERVICE% AppStderr "%LOGDIR%\worker.err.log"
42
+
43
+ REM Stop = send Ctrl-C (node SIGINT, graceful drain), wait 15000ms (>= the 5s docker-stop grace) before
44
+ REM nssm escalates to a hard kill.
45
+ nssm set %SERVICE% AppStopMethodConsole 15000
46
+
47
+ REM StartLimit analogue: pause 5000ms between restarts so a crash loop does not spin the provider bill
48
+ REM (mirrors StartLimitIntervalSec/StartLimitBurst + RestartSec in deploy/worker.service).
49
+ nssm set %SERVICE% AppThrottle 5000
50
+
51
+ REM Restart on a crash by default...
52
+ nssm set %SERVICE% AppExit Default Restart
53
+ REM ...but exit 2 is EXIT_POLICY: a determinate config/budget refusal, never retried. Do NOT restart it
54
+ REM (mirrors RestartPreventExitStatus=2 in deploy/worker.service).
55
+ nssm set %SERVICE% AppExit 2 Exit
56
+
57
+ echo Installed service "%SERVICE%". Start it with: nssm start %SERVICE%
58
+
59
+ endlocal
@@ -0,0 +1,36 @@
1
+ # UNTESTED EXAMPLE (DES-WORKER-ON-HOST) -- a starting point, not a shipped, verified unit. Adapt it.
2
+ #
3
+ # The receiver runs on the HOST as the public edge: it verifies GitHub deliveries and enqueues jobs.
4
+ # Requires node >=22.19.0 on PATH for User=pi -- or pin an absolute `ExecStart=/usr/bin/node ...`.
5
+ # This is a Linux/systemd unit. The worker ships cross-platform daemon artifacts under deploy/
6
+ # (com.pi-dispatch.worker.plist for launchd, nssm-install.cmd for Windows); receiver daemonization is
7
+ # out of scope, but a launchd/nssm receiver would follow the same .env-wrapper pattern those use.
8
+ #
9
+ # NAT / tunnel: the receiver binds `RECEIVER_BIND` (default 0.0.0.0) and must be reachable by GitHub's
10
+ # webhook delivery. On a home machine behind NAT, put it behind a tunnel (cloudflared / ngrok /
11
+ # tailscale funnel) or a reverse proxy with TLS -- do not port-forward it raw without one.
12
+ # `WEBHOOK_SECRET` is what authenticates deliveries; without a public URL GitHub cannot deliver.
13
+ #
14
+ # Env vars come from your `.env` (see `.env.example`) via EnvironmentFile -- never commit real secrets.
15
+ # WorkingDirectory / EnvironmentFile / User / node path below are PLACEHOLDERS: set them to wherever
16
+ # you cloned the repo and whoever owns it.
17
+
18
+ [Unit]
19
+ Description=pi-dispatch webhook receiver (public edge: verifies GitHub deliveries and enqueues jobs)
20
+ After=network-online.target
21
+ Wants=network-online.target
22
+
23
+ [Service]
24
+ Type=simple
25
+ User=pi
26
+ WorkingDirectory=/opt/pi-dispatch
27
+ EnvironmentFile=/opt/pi-dispatch/.env
28
+ ExecStart=/usr/bin/node receiver/src/start.mjs
29
+ Restart=on-failure
30
+ RestartSec=5
31
+ # The receiver handles SIGTERM: it closes the HTTP server and the queue connection, then exits.
32
+ KillSignal=SIGTERM
33
+ TimeoutStopSec=30
34
+
35
+ [Install]
36
+ WantedBy=multi-user.target
@@ -0,0 +1,50 @@
1
+ @echo off
2
+ REM UNTESTED EXAMPLE -- a starting point for a Windows service, not a shipped, verified unit. Adapt it.
3
+ REM
4
+ REM pi-dispatch launcher for Windows service managers (nssm; see deploy/nssm-install.cmd). Windows
5
+ REM services have no `.env` mechanism, so this wrapper loads `.env` from the repo root itself, then
6
+ REM launches node. It reads ONLY the declared `.env` (see `.env.example`), never the host user profile:
7
+ REM the container-boundary rules require an explicit, auditable variable set. Nothing here contains a
8
+ REM credential -- the secrets live in `.env`, which is gitignored and read at runtime.
9
+ REM
10
+ REM TRAP: inside pi, ANTHROPIC_OAUTH_TOKEN silently takes precedence over ANTHROPIC_API_KEY. Set exactly
11
+ REM one in `.env`.
12
+ REM
13
+ REM `.env` FORMAT for this loader: KEY=VALUE, one per line. Values MUST be UNQUOTED -- cmd's `set` keeps
14
+ REM surrounding quotes as part of the value. `eol=#` skips `#` comment lines; blank lines are ignored.
15
+ REM `tokens=1,* delims==` splits on the FIRST `=` only, so values containing `=` (base64, API keys)
16
+ REM survive intact.
17
+ REM
18
+ REM One worker per host (DES-CONCURRENCY-3): parallelism is PI_CONCURRENCY inside the single process, not
19
+ REM multiple services. Requires the AOF-enabled Valkey from deploy/docker-compose.yml.
20
+
21
+ setlocal
22
+
23
+ REM Resolve repo root relative to this script (deploy\ is one level down).
24
+ cd /d "%~dp0.." || exit /b 1
25
+
26
+ if not exist ".env" (
27
+ echo worker-env-wrapper: .env not found in "%CD%" 1>&2
28
+ exit /b 1
29
+ )
30
+
31
+ for /f "usebackq eol=# tokens=1,* delims==" %%A in (".env") do set "%%A=%%B"
32
+
33
+ REM One wrapper serves both daemons (see the .sh twin): no argument runs the worker; `receiver`
34
+ REM (passed by `pi-dispatch service --receiver`) runs the webhook receiver.
35
+ if "%~1"=="receiver" (
36
+ node receiver\src\start.mjs
37
+ ) else (
38
+ node worker\src\cli.mjs worker
39
+ )
40
+ set "RC=%ERRORLEVEL%"
41
+
42
+ REM Exit 2 is EXIT_POLICY (worker\src\exit-code.mjs): a determinate config/budget refusal. nssm's
43
+ REM `AppExit 2 Exit` already refuses to restart it, but converting to a clean 0 here keeps ANY service
44
+ REM manager pointed at this wrapper from relaunch-looping a refusal into a provider bill (mirrors the
45
+ REM .sh twin, which exists for launchd's KeepAlive that cannot exclude a single exit code).
46
+ if "%RC%"=="2" (
47
+ echo worker-env-wrapper: policy refusal, exit 2: not restarting; fix the config and start the service again 1>&2
48
+ exit /b 0
49
+ )
50
+ exit /b %RC%
@@ -0,0 +1,63 @@
1
+ #!/bin/sh
2
+ # pi-dispatch launcher for daemon managers that have NO EnvironmentFile mechanism. systemd reads
3
+ # `.env` for you via `EnvironmentFile=` (see deploy/worker.service); launchd (macOS) has no equivalent --
4
+ # a plist's ProgramArguments cannot name a `.env`. This wrapper closes that gap: launchd execs THIS
5
+ # script, which loads the explicit `.env` from the repo root and then runs node. Its exit-code
6
+ # conversion and signal forwarding are exercised under `sh` by worker/test/service.test.mjs.
7
+ #
8
+ # It sources ONLY the declared `.env` (see `.env.example`), never the host login shell: the
9
+ # container-boundary rules require an explicit, auditable variable set, not whatever the operator's
10
+ # profile happens to export. Nothing here contains a credential -- the secrets live in `.env`, which is
11
+ # gitignored and read at runtime.
12
+ #
13
+ # TRAP: inside pi, `ANTHROPIC_OAUTH_TOKEN` silently takes precedence over `ANTHROPIC_API_KEY`. Set exactly
14
+ # one in `.env`; this wrapper only ADDS the `.env` vars on top of the current environment, it does not
15
+ # clear a stray pre-existing one, so a leaked host `ANTHROPIC_OAUTH_TOKEN` would still win.
16
+ #
17
+ # One worker per host (DES-CONCURRENCY-3): parallelism is PI_CONCURRENCY inside the single process, not
18
+ # multiple daemons. Requires the AOF-enabled Valkey from deploy/docker-compose.yml.
19
+
20
+ # Resolve repo root relative to this script (deploy/ is one level down).
21
+ cd "$(dirname "$0")/.." || exit 1
22
+ if [ ! -f .env ]; then echo "worker-env-wrapper: .env not found in $(pwd)" >&2; exit 1; fi
23
+ set -a; . ./.env; set +a
24
+
25
+ # One wrapper serves both daemons, because the gap it closes (no EnvironmentFile under launchd/nssm)
26
+ # is identical for both: no argument runs the worker; `receiver` (passed by the derived receiver
27
+ # units `pi-dispatch service` renders) runs the webhook receiver.
28
+ if [ "$1" = "receiver" ]; then
29
+ set -- node receiver/src/start.mjs
30
+ else
31
+ set -- node worker/src/cli.mjs worker
32
+ fi
33
+
34
+ # `exec` is deliberately GONE here (it used to hand this shell's pid straight to node): intercepting
35
+ # the exit code needs a parent still alive after node exits. launchd's KeepAlive/SuccessfulExit=false
36
+ # relaunches ANY nonzero exit -- including EXIT_POLICY (2, worker/src/exit-code.mjs), the determinate
37
+ # config/budget refusal that systemd (RestartPreventExitStatus=2) and nssm (AppExit 2 Exit) both
38
+ # deliberately never retry. A relaunch loop against a paid provider is a bill, so the conversion at
39
+ # the bottom turns exit 2 into the clean exit KeepAlive leaves stopped.
40
+ #
41
+ # SIGTERM still reaches node without exec: the trap forwards TERM/INT to the child, and `wait` (unlike
42
+ # a foreground command in sh, which blocks trap delivery) is interruptible by a trapped signal, so the
43
+ # forwarding is immediate and node gets its full graceful drain.
44
+ signaled=0
45
+ trap 'signaled=1; kill -TERM "$child" 2>/dev/null' TERM INT
46
+ "$@" &
47
+ child=$!
48
+ wait "$child"
49
+ rc=$?
50
+ # The double wait is load-bearing: a trapped signal interrupts the FIRST wait early (rc = 128+signum)
51
+ # while node is still draining, so a SECOND wait is needed to collect node's real exit code. Guarded
52
+ # on both conditions so a normal exit never waits twice -- re-waiting on an already-reaped pid would
53
+ # read as 127, clobbering the true code.
54
+ if [ "$signaled" -eq 1 ] && [ "$rc" -ge 128 ]; then
55
+ wait "$child"
56
+ rc=$?
57
+ fi
58
+
59
+ if [ "$rc" -eq 2 ]; then
60
+ echo "worker-env-wrapper: policy refusal (exit 2): not restarting; fix the config and start the service again" >&2
61
+ exit 0
62
+ fi
63
+ exit "$rc"
@@ -0,0 +1,55 @@
1
+ # TEMPLATE — verified structure (systemd-analyze); set the <PLACEHOLDER> values for your host.
2
+ #
3
+ # The worker runs on the HOST, not in a container (DES-WORKER-ON-HOST): it drives the `docker` CLI to
4
+ # launch one job container per job, and the CLI is what translates bind-mount paths cross-platform.
5
+ # Compose runs only Valkey; this unit runs the worker beside it.
6
+ #
7
+ # Requires node >=22.19.0 (worker/package.json engines) on PATH for User=pi -- or pin an absolute
8
+ # `ExecStart=/usr/bin/node ...` if PATH is not reliable under systemd. This is a Linux/systemd unit;
9
+ # the launchd (macOS) equivalent is deploy/com.pi-dispatch.worker.plist and the Windows (nssm) one is
10
+ # deploy/nssm-install.cmd.
11
+ #
12
+ # Env vars come from your `.env` (see `.env.example`) via EnvironmentFile -- never commit real secrets.
13
+ # WorkingDirectory / EnvironmentFile / User / node path below are PLACEHOLDERS: set them to wherever
14
+ # you cloned the repo and whoever owns it.
15
+ #
16
+ # PI_LOGS_DIR (run-history records; default OS-temp /pi-dispatch/logs) is created and written by the
17
+ # worker at boot, so it must be writable by User= (pi) -- do NOT pre-create it as root, or the non-root
18
+ # worker hits EACCES at boot. Set via `.env` (EnvironmentFile), no unit change needed; the default path
19
+ # avoids colliding with the daemon's own logs/worker.out.log.
20
+ #
21
+ # PI_SETTINGS_FILE is the runtime-tunable settings overlay (default under OS temp, which may be wiped on
22
+ # reboot) -- point it at a durable path in production. Set via `.env` (EnvironmentFile), no unit change
23
+ # needed; it is worker-owned and never belongs in the container env allowlist.
24
+
25
+ [Unit]
26
+ Description=pi-dispatch worker (drains the job queue on the host; launches job containers via docker)
27
+ After=network-online.target docker.service
28
+ Wants=network-online.target
29
+ # Valkey must be reachable (docker compose -f deploy/docker-compose.yml up -d), but it is a separate
30
+ # unit/container -- not ordered here since it may be remote.
31
+ # Crash-loop bound: at most StartLimitBurst restarts within StartLimitIntervalSec, then systemd stops
32
+ # trying. A restart loop against a paid provider is a bill, not just log noise. Pairs with
33
+ # Restart=on-failure below.
34
+ StartLimitIntervalSec=60
35
+ StartLimitBurst=5
36
+
37
+ [Service]
38
+ Type=simple
39
+ User=pi
40
+ WorkingDirectory=/opt/pi-dispatch
41
+ EnvironmentFile=/opt/pi-dispatch/.env
42
+ ExecStart=/usr/bin/node worker/src/cli.mjs worker
43
+ Restart=on-failure
44
+ RestartSec=5
45
+ # Exit 2 is EXIT_POLICY (worker/src/exit-code.mjs): a determinate config/budget refusal, not infra.
46
+ # Never restart it -- retrying pays again for the same broken config. Only infra failures are worth a
47
+ # restart, and Restart=on-failure already covers those.
48
+ RestartPreventExitStatus=2
49
+ # The worker handles SIGTERM: it stops accepting new jobs and lets the in-flight container finish or
50
+ # abort cleanly. Give it room before SIGKILL.
51
+ KillSignal=SIGTERM
52
+ TimeoutStopSec=30
53
+
54
+ [Install]
55
+ WantedBy=multi-user.target
package/package.json ADDED
@@ -0,0 +1,83 @@
1
+ {
2
+ "name": "@edgehero/pi-dispatch",
3
+ "version": "0.1.0",
4
+ "type": "module",
5
+ "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
6
+ "keywords": [
7
+ "pi",
8
+ "pi-coding-agent",
9
+ "coding-agent",
10
+ "ai-agent",
11
+ "autonomous-agents",
12
+ "self-hosted",
13
+ "github",
14
+ "gitlab",
15
+ "forgejo",
16
+ "gitea",
17
+ "azure-devops",
18
+ "job-queue",
19
+ "bullmq",
20
+ "cron",
21
+ "docker",
22
+ "cli"
23
+ ],
24
+ "license": "MIT",
25
+ "author": "Rob Boerman",
26
+ "homepage": "https://github.com/edgehero/pi-dispatch/tree/main/worker",
27
+ "repository": {
28
+ "type": "git",
29
+ "url": "git+https://github.com/edgehero/pi-dispatch.git",
30
+ "directory": "worker"
31
+ },
32
+ "bugs": "https://github.com/edgehero/pi-dispatch/issues",
33
+ "files": [
34
+ "src",
35
+ ".env.example",
36
+ "deploy"
37
+ ],
38
+ "main": "src/index.mjs",
39
+ "bin": {
40
+ "pi-dispatch": "src/cli.mjs"
41
+ },
42
+ "exports": {
43
+ ".": "./src/index.mjs",
44
+ "./config": "./src/config.mjs",
45
+ "./exit-code": "./src/exit-code.mjs",
46
+ "./flow-gate": "./src/flow-gate.mjs",
47
+ "./git-dirty": "./src/git-dirty.mjs",
48
+ "./queue": "./src/queue.mjs",
49
+ "./connection": "./src/connection.mjs",
50
+ "./job-id": "./src/job-id.mjs",
51
+ "./forges": "./src/forges.mjs",
52
+ "./triggers": "./src/triggers.mjs",
53
+ "./packages": "./src/packages.mjs",
54
+ "./pause-windows": "./src/pause-windows.mjs",
55
+ "./identity": "./src/identity.mjs",
56
+ "./gitlab-identity": "./src/gitlab-identity.mjs",
57
+ "./forgejo-identity": "./src/forgejo-identity.mjs",
58
+ "./azure-identity": "./src/azure-identity.mjs",
59
+ "./get-token": "./src/get-token.mjs",
60
+ "./runtime-settings": "./src/runtime-settings.mjs",
61
+ "./run-history": "./src/run-history.mjs",
62
+ "./sandbox": "./src/sandbox.mjs",
63
+ "./sandbox-store": "./src/sandbox-store.mjs",
64
+ "./subscriptions": "./src/subscriptions.mjs",
65
+ "./budget": "./src/budget.mjs",
66
+ "./pricing": "./src/pricing.mjs",
67
+ "./scheduler-stall-guard": "./src/scheduler-stall-guard.mjs"
68
+ },
69
+ "engines": {
70
+ "node": ">=22.19.0"
71
+ },
72
+ "scripts": {
73
+ "test": "node --test \"test/*.test.mjs\"",
74
+ "start": "node src/cli.mjs worker"
75
+ },
76
+ "dependencies": {
77
+ "@earendil-works/pi-ai": "0.80.7",
78
+ "@octokit/auth-app": "8.2.0",
79
+ "@octokit/rest": "22.0.1",
80
+ "bullmq": "5.80.4",
81
+ "ioredis": "5.11.1"
82
+ }
83
+ }
@@ -0,0 +1,61 @@
1
+ /**
2
+ * Azure DevOps authentication for the worker, yielding the identical `{ mintToken, selfId, source }` shape
3
+ * every other forge does.
4
+ *
5
+ * CONST-TOKEN-SCOPED-PER-JOB, and the interesting thing here is the SYMMETRY with Forgejo: the two new
6
+ * forges fail OPPOSITE halves of this constraint.
7
+ *
8
+ * - short-lived YES, and better than Forgejo's or GitLab's: an Azure PAT carries a real expiry
9
+ * the operator chooses, up to a year, and the organization can cap it by policy.
10
+ * - repo-scoped NO, not by the token. PAT scopes are ORGANIZATION-wide -- `vso.code_write` grants
11
+ * write to every repository in the org, and there is no per-repository scope to
12
+ * select. The bound has to come from somewhere else.
13
+ * - host-held YES; it lives in the worker's env and reaches a container only as an env value.
14
+ * - env-injected YES; never written to /workspace, .git/config, argv, or a log.
15
+ * - not merge-capable YES in practice; branch policies are the barrier and no completion API is called.
16
+ * - minimally-permissioned Partly. The scopes are coarse, but they ARE separable: `vso.code_write` does
17
+ * not imply `vso.work_write`, unlike GitLab's all-or-nothing `api`.
18
+ *
19
+ * SO THE OPERATOR OBLIGATION IS DIFFERENT IN KIND, and the docs have to say which. On GitLab and Forgejo it
20
+ * is "rotate it, because nothing expires it". On Azure the expiry is fine and the SCOPE is the gap: the
21
+ * token must belong to a dedicated identity whose per-repository permissions are set in Project Settings,
22
+ * because that identity's own access -- not the token's scopes -- is what bounds the blast radius.
23
+ *
24
+ * All side-effecting collaborators are INJECTED, so the module is testable offline with no Azure.
25
+ */
26
+
27
+ import { configError } from "./config.mjs";
28
+ import { resolveAzureSelfId } from "./azure-identity.mjs";
29
+
30
+ /** Build the auth surface for `cfg = { source, orgUrl, tokenVar }`. Fails CLOSED at construction. */
31
+ export async function makeAzureAuth(cfg, deps = {}) {
32
+ const { env = process.env, fetchFn = fetch } = deps;
33
+ const source = cfg?.source;
34
+ if (source !== "pat") {
35
+ throw configError(`makeAzureAuth: unknown or missing source: ${JSON.stringify(source)} (only "pat" is supported -- Azure DevOps has no App or installation-token equivalent)`);
36
+ }
37
+
38
+ const tokenVar = cfg.tokenVar ?? "AZURE_TOKEN";
39
+ const token = requireToken(env[tokenVar], tokenVar);
40
+ const selfId = await resolveAzureSelfId({ orgUrl: cfg.orgUrl, token, fetchFn });
41
+
42
+ // Ignores the job by design: one operator-supplied token serves every project this deployment services,
43
+ // exactly as the other forges' pat sources do. The parameter exists so the shape matches.
44
+ const mintToken = async () => requireToken(token, tokenVar);
45
+ return { mintToken, selfId, source };
46
+ }
47
+
48
+ /**
49
+ * The money-hole invariant, mirrored from get-token.mjs: return a trimmed non-empty token, or throw.
50
+ *
51
+ * An empty credential would reach env-allowlist's truthiness check as falsy, the token would be OMITTED
52
+ * from the container env entirely, and the job would run anonymously -- a silent, paid, useless run rather
53
+ * than an error.
54
+ */
55
+ function requireToken(raw, what) {
56
+ const token = typeof raw === "string" ? raw.trim() : "";
57
+ if (token === "") {
58
+ throw configError(`${what} is empty or unset; refusing to hand a job an empty Azure DevOps credential`);
59
+ }
60
+ return token;
61
+ }