@wardby/cli 0.2.0 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (157) hide show
  1. package/README.md +142 -67
  2. package/dist/claude-coding-worker/driver.d.ts +1 -0
  3. package/dist/claude-coding-worker/driver.js +4 -1
  4. package/dist/claude-coding-worker/main.js +3 -2
  5. package/dist/coding/protocol.d.ts +28 -3
  6. package/dist/coding/protocol.js +53 -2
  7. package/dist/coding-worker/artifact.d.ts +12 -0
  8. package/dist/coding-worker/artifact.js +48 -6
  9. package/dist/coding-worker/driver.d.ts +1 -0
  10. package/dist/coding-worker/driver.js +4 -1
  11. package/dist/coding-worker/errors.d.ts +7 -0
  12. package/dist/coding-worker/errors.js +18 -0
  13. package/dist/coding-worker/main.js +3 -2
  14. package/dist/core/budget-groups.d.ts +1 -1
  15. package/dist/core/coding-queue.d.ts +1 -1
  16. package/dist/core/datastores.d.ts +1 -1
  17. package/dist/core/db.d.ts +36 -2
  18. package/dist/core/db.js +69 -2
  19. package/dist/core/dispatch.d.ts +16 -1
  20. package/dist/core/dispatch.js +33 -5
  21. package/dist/core/http-runtime.js +3 -4
  22. package/dist/core/lease.d.ts +1 -1
  23. package/dist/core/private-directory.d.ts +20 -0
  24. package/dist/core/private-directory.js +36 -0
  25. package/dist/core/reconciler.d.ts +1 -1
  26. package/dist/core/runner.d.ts +1 -1
  27. package/dist/core/scheduler.d.ts +1 -1
  28. package/dist/core/secrets.d.ts +1 -1
  29. package/dist/core/subagent-memory-tools.d.ts +1 -1
  30. package/dist/core/webhooks.d.ts +1 -1
  31. package/dist/generated/prisma/browser.d.ts +228 -0
  32. package/dist/generated/prisma/browser.js +17 -0
  33. package/dist/generated/prisma/client.d.ts +247 -0
  34. package/dist/generated/prisma/client.js +34 -0
  35. package/dist/generated/prisma/commonInputTypes.d.ts +809 -0
  36. package/dist/generated/prisma/commonInputTypes.js +10 -0
  37. package/dist/generated/prisma/enums.d.ts +53 -0
  38. package/dist/generated/prisma/enums.js +54 -0
  39. package/dist/generated/prisma/internal/class.d.ts +462 -0
  40. package/dist/generated/prisma/internal/class.js +49 -0
  41. package/dist/generated/prisma/internal/prismaNamespace.d.ts +3325 -0
  42. package/dist/generated/prisma/internal/prismaNamespace.js +455 -0
  43. package/dist/generated/prisma/internal/prismaNamespaceBrowser.d.ts +449 -0
  44. package/dist/generated/prisma/internal/prismaNamespaceBrowser.js +426 -0
  45. package/dist/generated/prisma/models/Agent.d.ts +3267 -0
  46. package/dist/generated/prisma/models/Agent.js +1 -0
  47. package/dist/generated/prisma/models/AgentDatastore.d.ts +1210 -0
  48. package/dist/generated/prisma/models/AgentDatastore.js +1 -0
  49. package/dist/generated/prisma/models/AgentMemory.d.ts +986 -0
  50. package/dist/generated/prisma/models/AgentMemory.js +1 -0
  51. package/dist/generated/prisma/models/AgentSecret.d.ts +1212 -0
  52. package/dist/generated/prisma/models/AgentSecret.js +1 -0
  53. package/dist/generated/prisma/models/AgentSubAgent.d.ts +1217 -0
  54. package/dist/generated/prisma/models/AgentSubAgent.js +1 -0
  55. package/dist/generated/prisma/models/AgentTool.d.ts +1309 -0
  56. package/dist/generated/prisma/models/AgentTool.js +1 -0
  57. package/dist/generated/prisma/models/AuthFormChallenge.d.ts +1298 -0
  58. package/dist/generated/prisma/models/AuthFormChallenge.js +1 -0
  59. package/dist/generated/prisma/models/AuthLoginKey.d.ts +1245 -0
  60. package/dist/generated/prisma/models/AuthLoginKey.js +1 -0
  61. package/dist/generated/prisma/models/AuthRateLimit.d.ts +979 -0
  62. package/dist/generated/prisma/models/AuthRateLimit.js +1 -0
  63. package/dist/generated/prisma/models/AuthSession.d.ts +1376 -0
  64. package/dist/generated/prisma/models/AuthSession.js +1 -0
  65. package/dist/generated/prisma/models/AuthUser.d.ts +1645 -0
  66. package/dist/generated/prisma/models/AuthUser.js +1 -0
  67. package/dist/generated/prisma/models/BudgetGroup.d.ts +1548 -0
  68. package/dist/generated/prisma/models/BudgetGroup.js +1 -0
  69. package/dist/generated/prisma/models/CodingAgentProfile.d.ts +1430 -0
  70. package/dist/generated/prisma/models/CodingAgentProfile.js +1 -0
  71. package/dist/generated/prisma/models/CodingProxyRequest.d.ts +1749 -0
  72. package/dist/generated/prisma/models/CodingProxyRequest.js +1 -0
  73. package/dist/generated/prisma/models/CodingProxySession.d.ts +1505 -0
  74. package/dist/generated/prisma/models/CodingProxySession.js +1 -0
  75. package/dist/generated/prisma/models/CodingRun.d.ts +2402 -0
  76. package/dist/generated/prisma/models/CodingRun.js +1 -0
  77. package/dist/generated/prisma/models/Datastore.d.ts +1434 -0
  78. package/dist/generated/prisma/models/Datastore.js +1 -0
  79. package/dist/generated/prisma/models/DatastoreEntry.d.ts +1314 -0
  80. package/dist/generated/prisma/models/DatastoreEntry.js +1 -0
  81. package/dist/generated/prisma/models/OAuthAuthorizationCode.d.ts +1526 -0
  82. package/dist/generated/prisma/models/OAuthAuthorizationCode.js +1 -0
  83. package/dist/generated/prisma/models/OAuthAuthorizationRequest.d.ts +1350 -0
  84. package/dist/generated/prisma/models/OAuthAuthorizationRequest.js +1 -0
  85. package/dist/generated/prisma/models/OAuthClient.d.ts +1289 -0
  86. package/dist/generated/prisma/models/OAuthClient.js +1 -0
  87. package/dist/generated/prisma/models/OAuthFamily.d.ts +1540 -0
  88. package/dist/generated/prisma/models/OAuthFamily.js +1 -0
  89. package/dist/generated/prisma/models/OAuthGrant.d.ts +1280 -0
  90. package/dist/generated/prisma/models/OAuthGrant.js +1 -0
  91. package/dist/generated/prisma/models/Principal.d.ts +1790 -0
  92. package/dist/generated/prisma/models/Principal.js +1 -0
  93. package/dist/generated/prisma/models/Run.d.ts +2281 -0
  94. package/dist/generated/prisma/models/Run.js +1 -0
  95. package/dist/generated/prisma/models/SchedulerLease.d.ts +970 -0
  96. package/dist/generated/prisma/models/SchedulerLease.js +1 -0
  97. package/dist/generated/prisma/models/Secret.d.ts +1399 -0
  98. package/dist/generated/prisma/models/Secret.js +1 -0
  99. package/dist/generated/prisma/models/SecretElicitationOutcome.d.ts +973 -0
  100. package/dist/generated/prisma/models/SecretElicitationOutcome.js +1 -0
  101. package/dist/generated/prisma/models/Task.d.ts +1176 -0
  102. package/dist/generated/prisma/models/Task.js +1 -0
  103. package/dist/generated/prisma/models/Tool.d.ts +1478 -0
  104. package/dist/generated/prisma/models/Tool.js +1 -0
  105. package/dist/generated/prisma/models/Webhook.d.ts +1386 -0
  106. package/dist/generated/prisma/models/Webhook.js +1 -0
  107. package/dist/generated/prisma/models.d.ts +32 -0
  108. package/dist/generated/prisma/models.js +1 -0
  109. package/dist/import/create.d.ts +1 -1
  110. package/dist/import/create.js +1 -1
  111. package/dist/import/index.d.ts +1 -1
  112. package/dist/mcp/auth/ownership.d.ts +19 -19
  113. package/dist/mcp/auth/principal.d.ts +1 -1
  114. package/dist/mcp/auth/resource-server.d.ts +1 -1
  115. package/dist/mcp/auth/self-hosted/browser.js +20 -6
  116. package/dist/mcp/auth/self-hosted/cli.d.ts +1 -1
  117. package/dist/mcp/auth/self-hosted/credentials.d.ts +1 -1
  118. package/dist/mcp/auth/self-hosted/rate-limit.d.ts +1 -1
  119. package/dist/mcp/auth/self-hosted/session.d.ts +1 -1
  120. package/dist/mcp/context.d.ts +1 -1
  121. package/dist/mcp/server.d.ts +1 -1
  122. package/dist/mcp/shared/page-style.d.ts +16 -0
  123. package/dist/mcp/shared/page-style.js +129 -0
  124. package/dist/mcp/tasks/manager.d.ts +1 -1
  125. package/dist/mcp/tools/agents.js +1 -1
  126. package/dist/mcp/tools/datastore.js +1 -1
  127. package/dist/mcp/tools/scheduling.js +1 -1
  128. package/dist/mcp/tools/secret-elicitation-form.d.ts +1 -1
  129. package/dist/mcp/tools/secret-elicitation-form.js +10 -28
  130. package/dist/mcp/tools/secret-elicitation.d.ts +1 -1
  131. package/dist/mcp/tools/subagents.js +1 -1
  132. package/dist/mcp/tools/tools.js +1 -1
  133. package/dist/mcp/transport/streamable-http.d.ts +1 -1
  134. package/dist/mcp/unattended-schedules.d.ts +1 -1
  135. package/dist/mcp/webhooks/ingress.d.ts +1 -1
  136. package/dist/providers/auth/index.d.ts +1 -1
  137. package/dist/providers/auth/self-hosted.d.ts +2 -2
  138. package/dist/providers/coding-proxy/prisma-ledger.d.ts +1 -1
  139. package/dist/providers/coding-proxy/runtime.d.ts +1 -1
  140. package/dist/providers/datastore/postgres.d.ts +1 -1
  141. package/dist/providers/executor/build.d.ts +1 -1
  142. package/dist/providers/executor/composition.d.ts +1 -1
  143. package/dist/providers/executor/container.d.ts +1 -1
  144. package/dist/providers/executor/container.js +10 -3
  145. package/dist/providers/executor/dbos.d.ts +1 -1
  146. package/dist/providers/jobs/docker.d.ts +11 -0
  147. package/dist/providers/jobs/docker.js +34 -18
  148. package/dist/providers/jobs/kubernetes.js +5 -10
  149. package/dist/providers/jobs/types.d.ts +2 -0
  150. package/dist/providers/memory/postgres.d.ts +1 -1
  151. package/dist/providers/vcs/git.js +2 -1
  152. package/dist/quickstart/index.js +22 -31
  153. package/dist/quickstart/migrate.d.ts +23 -0
  154. package/dist/quickstart/migrate.js +125 -0
  155. package/package.json +20 -10
  156. package/prisma/migrate.config.mjs +20 -0
  157. package/prisma/schema.prisma +10 -2
package/README.md CHANGED
@@ -2,16 +2,17 @@
2
2
  <img align="left" hspace="24" src="https://raw.githubusercontent.com/wardby/wardby/main/docs/assets/brand/wardby-mascot.png" alt="Wardby guardian robot protecting an agent budget" width="280">
3
3
  <h3><big><big>Wardby</big></big> <small><em>(pronounced&nbsp;“WARD&#8209;bee”)</em></small></h3>
4
4
  <h3>Autonomous agents, bounded by design.</h3>
5
- <p><strong>Most agent runners focus on helping a model complete a task. Wardby is the self-hosted control plane that decides whether the task should run, limits what it can access, and returns a reviewable outcome governed by explicit policy.</strong></p>
5
+ <p><strong>Agent runners help a model complete a task. Wardby is the self-hosted control plane that governs the work around it: whether it may run, what it may access, what it may spend, and what reviewable outcome it may produce.</strong></p>
6
6
  <br clear="left">
7
7
  <p><strong>Budgets are enforced before spend: every model request must fit within a hard run limit before it reaches the provider.</strong></p>
8
+ <p><strong>Bring your providers. Keep your infrastructure. Govern agents in one place.</strong></p>
8
9
  <p>
10
+ <a href="#get-started">Get started</a> ·
9
11
  <a href="#why-wardby">Why Wardby</a> ·
10
- <a href="#why-wardby-instead-of-another-agent-runner">Why it is different</a> ·
12
+ <a href="#where-wardby-fits">Where it fits</a> ·
11
13
  <a href="#a-full-cycle-agent-from-one-conversation">Full-cycle example</a> ·
12
14
  <a href="#host-it-in-your-cloud">Deployments</a> ·
13
15
  <a href="#bring-your-own-observability">Observability</a> ·
14
- <a href="https://github.com/wardby/wardby/blob/main/docs/getting-started.md">Getting started</a> ·
15
16
  <a href="#security-boundaries">Security</a>
16
17
  </p>
17
18
  </div>
@@ -37,50 +38,138 @@ there resolves against npmjs.com instead of this repository. -->
37
38
  > and local observability stack are implemented and tested. Production
38
39
  > readiness and cloud deployment coverage are still being expanded.
39
40
 
41
+ ## Get started
42
+
43
+ ### Run locally in five minutes
44
+
45
+ Requirements: Node.js 24 or newer, Docker, and an OpenAI or Anthropic API key.
46
+ You do not need to clone Wardby or install PostgreSQL.
47
+
48
+ 1. From the project where you want to use Wardby, run:
49
+
50
+ ```sh
51
+ npx --yes @wardby/cli@latest quickstart
52
+ ```
53
+
54
+ 2. Follow the prompts to choose a provider, start PostgreSQL, create a `$1`
55
+ demo agent, and optionally connect Codex or Claude Code through MCP.
56
+
57
+ 3. Verify the installation:
58
+
59
+ ```sh
60
+ npx --yes @wardby/cli@latest doctor
61
+ ```
62
+
63
+ Quickstart keeps credentials and local state under `.wardby/` in the current
64
+ project. Read the [local getting-started guide](https://github.com/wardby/wardby/blob/main/docs/getting-started.md)
65
+ for unattended setup, lifecycle commands, MCP configuration, and cleanup.
66
+
67
+ ### Develop Wardby from source
68
+
69
+ Clone the repository only when you want to contribute to Wardby, inspect its
70
+ deployment assets, or build images yourself:
71
+
72
+ ```sh
73
+ git clone https://github.com/wardby/wardby.git
74
+ cd wardby
75
+ npm ci
76
+ npm run build
77
+ npm test
78
+ ```
79
+
80
+ Continue with [CONTRIBUTING.md](https://github.com/wardby/wardby/blob/main/CONTRIBUTING.md)
81
+ for the development database, migrations, required checks, and contribution
82
+ expectations.
83
+
84
+ ### Deploy for a team
85
+
86
+ - Follow the [GKE guide](https://github.com/wardby/wardby/blob/main/docs/getting-started-gke.md)
87
+ for the supported Google Cloud reference deployment.
88
+ - Start from the [portable production boundary](https://github.com/wardby/wardby/blob/main/deploy/production/README.md)
89
+ for another cloud, VM, or container platform.
90
+ - Follow [Bring your own identity provider](https://github.com/wardby/wardby/blob/main/docs/getting-started-identity-provider.md)
91
+ to protect remote MCP with your existing OAuth/OIDC provider.
92
+
40
93
  ## Why Wardby
41
94
 
42
95
  AI agents are easy to demo and harder to operate. Once an agent can spend
43
96
  money, use credentials, change a repository, or run without a person watching,
44
97
  teams need more than a prompt and a cron job.
45
98
 
99
+ ### One control plane instead of scattered automation
100
+
101
+ Teams often begin with Claude Code or Codex on developer machines,
102
+ repository-specific automation, and separate provider dashboards. As adoption
103
+ grows, agent definitions, ownership, credentials, budgets, schedules, and run
104
+ history become fragmented across tools and repositories.
105
+
106
+ Wardby gives every Wardby-managed agent one durable operational identity: who
107
+ owns it, why it exists, what it may access, when it runs, what it may spend,
108
+ and what outcome it produced.
109
+
46
110
  Wardby provides the control plane around the model:
47
111
 
48
- | For engineering leaders | For developers |
49
- | --------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------ |
50
- | Put explicit cost, ownership, and approval boundaries around agent work. | Create and manage agents from Claude or Codex through MCP. |
51
- | Keep execution, data, and credentials in infrastructure your team controls. | Choose OpenAI, Anthropic, or Bedrock-backed models per agent. |
52
- | Turn one-off experiments into scheduled, observable operating processes. | Attach scoped tools, secrets, schedules, memory, and sub-agents. |
53
- | Keep approval and action authority explicit for consequential outcomes. | Run optional coding tasks in isolated Codex or Claude Code workers that produce draft PRs. |
112
+ | For engineering leaders | For developers |
113
+ | ---------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------ |
114
+ | Keep an inventory of owned agents, purposes, schedules, limits, and results. | Create and manage agents from Claude or Codex through MCP. |
115
+ | See reserved budget and actual usage across Wardby-managed agents. | Choose OpenAI, Anthropic, or Bedrock-backed models per agent. |
116
+ | Keep execution, data, and credentials in infrastructure your team controls. | Switch approved models or builders without rewriting the governance contract. |
117
+ | Turn one-off experiments into scheduled, observable operating processes. | Attach scoped tools, secrets, schedules, memory, and sub-agents. |
118
+ | Keep approval and action authority explicit for consequential outcomes. | Run optional coding tasks in isolated Codex or Claude Code workers that produce draft PRs. |
54
119
 
55
120
  The result is not another autonomous black box. It is a way to make agent work
56
121
  repeatable, bounded, inspectable, and reviewable.
57
122
 
58
- ## Why Wardby instead of another agent runner?
123
+ ## Where Wardby fits
59
124
 
60
125
  Agent tools solve different layers of the problem. Wardby does not need to
61
126
  replace them: it provides the self-hosted operating boundary around agents and
62
127
  the work they perform.
63
128
 
64
- | Category | What it primarily helps you do | What Wardby adds |
65
- | -------------------------- | ----------------------------------------------------------- | ------------------------------------------------------------------------------------------ |
66
- | **Agent frameworks** | Build reasoning loops, tool calls, and multi-agent logic. | Persistent ownership, schedules, budgets, credentials, run history, and lifecycle control. |
67
- | **Coding agents** | Plan, edit, and test code for an interactive task. | Isolated managed workers, admission-time budgets, scoped access, and optional draft PRs. |
68
- | **Hosted agent platforms** | Start quickly on infrastructure operated by another vendor. | A control plane, data, credentials, and execution boundary you can host in your own cloud. |
69
- | **Workflow orchestrators** | Make application jobs durable, retryable, and observable. | Agent-specific policy, model usage, capabilities, budgets, and MCP-native operations. |
129
+ | Category | What it primarily helps you do | What Wardby adds |
130
+ | --------------------------- | ----------------------------------------------------------- | ------------------------------------------------------------------------------------------ |
131
+ | **Repository automation** | Run jobs for one repository in response to delivery events. | One agent catalog, budget model, and policy boundary across repositories and triggers. |
132
+ | **Agent frameworks** | Build reasoning loops, tool calls, and multi-agent logic. | Persistent ownership, schedules, budgets, credentials, run history, and lifecycle control. |
133
+ | **Coding agents** | Plan, edit, and test code for an interactive task. | Isolated managed workers, admission-time budgets, scoped access, and optional draft PRs. |
134
+ | **Provider dashboards** | Report usage and cost within one model provider. | Agent-owned budgets, capabilities, runs, and outcomes across approved providers. |
135
+ | **LLM gateways** | Route model requests, manage keys, and enforce quotas. | Budgets tied to named agents, owned runs, capabilities, schedules, and outcomes. |
136
+ | **Observability platforms** | Trace requests, evaluate quality, and explain cost. | Admission and execution controls applied before work occurs, plus a durable agent catalog. |
137
+ | **Vendor control planes** | Govern agents inside one provider or cloud ecosystem. | A self-hosted boundary spanning approved providers, builders, repositories, and clouds. |
138
+ | **Workflow orchestrators** | Make application jobs durable, retryable, and observable. | Agent-specific policy, model usage, capabilities, budgets, and MCP-native operations. |
139
+
140
+ The distinction is governed work, not just model calls or execution. Rather
141
+ than assembling a gateway, scheduler, agent catalog, budget service,
142
+ capability registry, and run database, Wardby provides one operating boundary.
143
+ It reserves spend before a run starts, grants only assigned capabilities,
144
+ records what happened, and keeps downstream action authority separate from the
145
+ worker that produced the result.
146
+
147
+ > **Scope:** Wardby's inventory and budget views cover work managed through
148
+ > Wardby. Provider reconciliation and unmanaged-agent discovery are required
149
+ > before those views can represent every AI agent or expense in an
150
+ > organization.
151
+
152
+ ## One operational contract, the full lifecycle
153
+
154
+ ![Wardby workflow: ask in Claude or Codex, define an agent through MCP, govern it in Wardby, execute it in isolation, and apply review policy to its outcome](https://raw.githubusercontent.com/wardby/wardby/main/docs/assets/wardby-workflow.svg)
70
155
 
71
- The distinction is control, not just execution. Wardby reserves spend before a
72
- run starts, grants only assigned capabilities, records what happened, and
73
- keeps downstream action authority separate from the worker that produced the
74
- result.
156
+ Claude and Codex are the operator experience. Wardby is the durable operating
157
+ boundary between intent and agent execution: it stores agent definitions,
158
+ triggers work, reserves budget, mediates tools and credentials, records
159
+ outcomes, and exposes run state through MCP. Nothing runs until identity,
160
+ policy, and available budget agree.
75
161
 
76
- ## One control plane, the full lifecycle
162
+ ### Governed shared state for agent teams
77
163
 
78
- ![Wardby workflow: ask in Claude or Codex, define an agent through MCP, govern it in Wardby, execute it in isolation, and apply review policy to its outcome](https://raw.githubusercontent.com/wardby/wardby/main/docs/assets/wardby-workflow.svg)
164
+ Named Wardby datastores let related agents exchange persistent structured data
165
+ without sharing an unrestricted database credential. Datastores are owned and
166
+ attached to agents explicitly; each sandboxed tool can reach only approved
167
+ bound names and key prefixes.
79
168
 
80
- Claude and Codex are the operator experience. Wardby is the durable system
81
- behind them: it stores agent definitions, triggers work, reserves budget,
82
- mediates tools and credentials, records outcomes, and exposes run state through
83
- MCP.
169
+ A planner can publish a feature plan, a builder can record implementation
170
+ state, and a reviewer or QA agent can add findings to the same governed
171
+ workspace. Datastores provide bounded coordination state, not a replacement
172
+ for authoritative source systems or searchable agent memory.
84
173
 
85
174
  ## A full-cycle agent from one conversation
86
175
 
@@ -107,13 +196,17 @@ Claude or Codex can translate that request into Wardby MCP operations such as
107
196
 
108
197
  The same agent can be updated, paused, triggered, inspected, or deleted from an
109
198
  MCP conversation. The CLI remains available as the bootstrap and operations
110
- floor.
199
+ floor. The result is autonomy with a receipt: an owned run with bounded spend,
200
+ assigned capabilities, durable status, and a reviewable outcome.
111
201
 
112
202
  ## Agent ecosystems you can compose
113
203
 
114
204
  Wardby provides lifecycle primitives rather than prescribing one fixed catalog.
115
205
  These are example systems a team can build and manage through MCP:
116
206
 
207
+ **Builders can vary. The controls do not.** Each system inherits the same
208
+ budget, capability, identity, evidence, and review contract.
209
+
117
210
  | Agent system | Typical cycle | Governed outcome |
118
211
  | ------------------------ | ------------------------------------------------------------------ | ------------------------------------ |
119
212
  | **Delivery pipeline** | Work request → plan → implementation → tests → review | Optional draft feature or bug-fix PR |
@@ -129,9 +222,15 @@ own model, tools, schedule, and run history.
129
222
 
130
223
  ## Host it in your cloud
131
224
 
225
+ **Bring your providers. Keep your infrastructure. Govern agents in one place.**
226
+
132
227
  The core does not import a cloud SDK. Jobs, model access, email, secrets,
133
228
  authentication, and object storage sit behind provider interfaces so operators
134
- can choose the infrastructure boundary that fits their environment.
229
+ can choose the infrastructure boundary that fits their environment. Models,
230
+ coding executors, cloud platforms, and observability systems can change
231
+ independently instead of defining the agent architecture. This reduces
232
+ control-plane lock-in without pretending that source control, model providers,
233
+ or monitoring systems disappear.
135
234
 
136
235
  | Target | Current support |
137
236
  | ------------------------------------ | ---------------------------------------------------------------------------------------------------------------- |
@@ -146,14 +245,10 @@ Start with [deployment targets](https://github.com/wardby/wardby/blob/main/deplo
146
245
  [GKE getting-started guide](https://github.com/wardby/wardby/blob/main/docs/getting-started-gke.md). Reference deployments are examples,
147
246
  not a requirement to use one vendor.
148
247
 
149
- **A complete deployment runs more than `wardby mcp`.** `mcp` serves the MCP
150
- surface; the scheduler that fires due agents and the reconciler that recovers
151
- orphaned runs live in `wardby scheduler`. Run **`wardby serve`** to get all three
152
- in one process. It is the container image's default command and what a
153
- single-container deployment (Cloud Run, a lone VM) should run. Alternatively,
154
- run `mcp` and `scheduler` as two processes, as `deploy/production/compose.yml`
155
- does. Running `mcp` alone logs a startup warning if enabled schedules exist
156
- that nothing will fire.
248
+ For a single-container deployment, run **`wardby serve`** to start MCP, the
249
+ scheduler, and reconciliation together. Split deployments can run `wardby mcp`
250
+ and `wardby scheduler` separately; `mcp` alone does not fire schedules. See the
251
+ deployment guides above for the complete process boundary.
157
252
 
158
253
  ## Bring your own observability
159
254
 
@@ -211,44 +306,24 @@ Review policy is designed to support designated agents as well as people.
211
306
  Agent approvals will be recorded as workflow evidence; operators will decide
212
307
  whether that evidence permits an automated action or still requires a person.
213
308
 
214
- ## Quickstart
309
+ ## Optional npm installation
215
310
 
216
- Requirements: Node.js 24 or newer, Docker, and one supported model-provider
217
- credential.
218
-
219
- ```sh
220
- npx --yes @wardby/cli@latest quickstart
221
- ```
222
-
223
- The guided command checks prerequisites, creates private project-local
224
- configuration, starts PostgreSQL in Docker, applies migrations, and offers to
225
- run a small agent with a `$1` maximum budget. It can also register Wardby as a
226
- local MCP server in Codex, Claude Code, or both.
227
-
228
- Read the [Getting started guide](https://github.com/wardby/wardby/blob/main/docs/getting-started.md) for unattended setup,
229
- MCP usage, lifecycle commands, and the boundary between the native demo and
230
- isolated coding agents. For a shared cloud installation, follow the full
231
- [GKE getting-started guide](https://github.com/wardby/wardby/blob/main/docs/getting-started-gke.md).
232
-
233
- ### Installing from npm
311
+ The `npx` quickstart above requires no installation. To pin Wardby as a local
312
+ project dependency instead:
234
313
 
235
314
  ```sh
236
315
  npm install @wardby/cli
237
316
  ```
238
317
 
239
- `npm audit` will report a high-severity advisory in `deepmerge-ts`
240
- ([GHSA-ggr8-5vv4-36mx](https://github.com/advisories/GHSA-ggr8-5vv4-36mx)). It
241
- is reached only through Prisma's CLI while it loads your own Prisma
242
- configuration, and Prisma has not patched it in the 6.x line. This repository
243
- clears it with an override, but npm does not apply a package's overrides to the
244
- projects that install it — so add the same entry to your own `package.json`:
245
-
246
- ```json
247
- "overrides": { "deepmerge-ts": "8.0.2" }
248
- ```
318
+ That install is audit-clean: the published package carries no Prisma CLI,
319
+ `@prisma/config`, `deepmerge-ts`, or `mysql2`, so `npm audit` reports nothing
320
+ and no `overrides` entry is needed in your own `package.json`.
249
321
 
250
- That override is tested against Prisma 6 and the CLI; the reasoning and the
251
- retirement plan are in [SR-009](https://github.com/wardby/wardby/blob/main/docs/security-deployment.md#images-and-dependency-exception).
322
+ `wardby quickstart` and `wardby doctor` separately fetch the pinned Prisma CLI
323
+ (`prisma@7.10.0`) on demand via `npx` to run migrations, so the machine running
324
+ them needs npm registry access at that moment. See
325
+ [Images and dependencies](https://github.com/wardby/wardby/blob/main/docs/security-deployment.md#images-and-dependencies)
326
+ for the repository's own dependency-override policy.
252
327
 
253
328
  ## What is implemented
254
329
 
@@ -20,6 +20,7 @@ export declare const CLAUDE_OUTPUT_JSON_SCHEMA: {
20
20
  };
21
21
  readonly tag: {
22
22
  readonly type: "string";
23
+ readonly description: "Optional short label shown in the pull request title, such as a ticket id. At most 32 characters: letters, digits, '.', '_', '/', '-', starting with a letter or digit, with no spaces. Example: \"add-jokes\". Omit it when there is no natural label.";
23
24
  };
24
25
  readonly tests: {
25
26
  readonly type: "array";
@@ -13,7 +13,10 @@ export const CLAUDE_OUTPUT_JSON_SCHEMA = {
13
13
  runId: { type: "string" },
14
14
  outcome: { type: "string", enum: ["changes_ready", "no_changes", "budget_exhausted"] },
15
15
  summary: { type: "string" },
16
- tag: { type: "string" },
16
+ tag: {
17
+ type: "string",
18
+ description: "Optional short label shown in the pull request title, such as a ticket id. At most 32 characters: letters, digits, '.', '_', '/', '-', starting with a letter or digit, with no spaces. Example: \"add-jokes\". Omit it when there is no natural label.",
19
+ },
17
20
  tests: {
18
21
  type: "array",
19
22
  maxItems: 64,
@@ -1,6 +1,6 @@
1
1
  #!/usr/bin/env node
2
2
  import { readCodingInput, writeCodingOutputAtomic } from "../coding-worker/artifact.js";
3
- import { safeWorkerErrorCode } from "../coding-worker/errors.js";
3
+ import { safeOutputIssues, safeWorkerErrorCode } from "../coding-worker/errors.js";
4
4
  import { runClaudeCodingWorker } from "./driver.js";
5
5
  import { createClaudeSdkQuery } from "./sdk.js";
6
6
  const INPUT_PATH = "/run/wardby/input/input.json";
@@ -36,6 +36,7 @@ catch (error) {
36
36
  : safeCode === "worker_failed"
37
37
  ? `worker_${stage}_failed`
38
38
  : safeCode;
39
- process.stderr.write(`${JSON.stringify({ error: code })}\n`);
39
+ const issues = code === "coding_output_invalid" ? safeOutputIssues(error) : undefined;
40
+ process.stderr.write(`${JSON.stringify({ error: code, ...(issues ? { issues } : {}) })}\n`);
40
41
  process.exitCode = controller.signal.aborted ? 143 : 1;
41
42
  }
@@ -6,10 +6,35 @@ export declare const MAX_CODING_SUMMARY_BYTES: number;
6
6
  export declare const MAX_CODING_TESTS = 64;
7
7
  export declare const MAX_CODING_TEST_COMMAND_BYTES: number;
8
8
  export declare const MAX_TAG_BYTES = 32;
9
+ /**
10
+ * A worker's output-schema failure, reduced to where and how it failed, never
11
+ * what the model wrote: `<path>:<zod issue code>`, where the path is dot-joined
12
+ * schema keys and array indices (`$` for the root). Unrecognized keys the model
13
+ * invented are carried only as the `unrecognized_keys` code, never by name.
14
+ */
15
+ export declare const MAX_CODING_OUTPUT_ISSUES = 8;
16
+ export declare const SAFE_CODING_OUTPUT_ISSUE: RegExp;
9
17
  export declare function normalizeGitHubRepository(value: string): string;
10
18
  export declare function normalizeGitRef(value: string): string;
11
19
  export declare const CodingBaseRefSchema: z.ZodEffects<z.ZodEffects<z.ZodString, string, string>, string, string>;
12
20
  export declare const CodingTaskOverrideSchema: z.ZodEffects<z.ZodEffects<z.ZodEffects<z.ZodString, string, string>, string, string>, string, string>;
21
+ /**
22
+ * A coding worker receives only its task text, so a coding agent's own
23
+ * instructions (its systemPrompt) travel inside that text, ahead of the
24
+ * request. Blank instructions leave the task unchanged. The combination must
25
+ * still fit MAX_CODING_TASK_BYTES; exceeding it is an error naming both parts
26
+ * rather than a silent truncation of either.
27
+ */
28
+ export declare function composeCodingTask(instructions: string | null | undefined, task: string): string;
29
+ /**
30
+ * The tag only labels a pull request title, so a model's malformed tag must not
31
+ * sink an otherwise finished run. A tag that already passes is kept as-is; any
32
+ * other string becomes a lowercase slug (runs of disallowed characters to "-",
33
+ * no leading or trailing punctuation, at most MAX_TAG_BYTES), and one with
34
+ * nothing usable left, or a non-string, becomes null. GitHub finalization still
35
+ * re-validates whatever survives.
36
+ */
37
+ export declare function normalizeCodingTag(value: unknown): string | null | undefined;
13
38
  export declare const CodingTaskInputSchema: z.ZodEffects<z.ZodObject<{
14
39
  schemaVersion: z.ZodLiteral<1>;
15
40
  runId: z.ZodEffects<z.ZodEffects<z.ZodEffects<z.ZodEffects<z.ZodString, string, string>, string, string>, string, string>, string, string>;
@@ -103,7 +128,7 @@ export declare const CodingAgentOutputSchema: z.ZodEffects<z.ZodObject<{
103
128
  command: string;
104
129
  outcome: "passed" | "failed" | "skipped";
105
130
  }>, "many">;
106
- tag: z.ZodEffects<z.ZodOptional<z.ZodNullable<z.ZodString>>, string | undefined, string | null | undefined>;
131
+ tag: z.ZodEffects<z.ZodEffects<z.ZodOptional<z.ZodNullable<z.ZodString>>, string | undefined, string | null | undefined>, string | undefined, unknown>;
107
132
  }, "strict", z.ZodTypeAny, {
108
133
  outcome: "changes_ready" | "no_changes" | "budget_exhausted";
109
134
  schemaVersion: 1;
@@ -123,7 +148,7 @@ export declare const CodingAgentOutputSchema: z.ZodEffects<z.ZodObject<{
123
148
  command: string;
124
149
  outcome: "passed" | "failed" | "skipped";
125
150
  }[];
126
- tag?: string | null | undefined;
151
+ tag?: unknown;
127
152
  }>, {
128
153
  tag?: string | undefined;
129
154
  summary: string;
@@ -143,7 +168,7 @@ export declare const CodingAgentOutputSchema: z.ZodEffects<z.ZodObject<{
143
168
  command: string;
144
169
  outcome: "passed" | "failed" | "skipped";
145
170
  }[];
146
- tag?: string | null | undefined;
171
+ tag?: unknown;
147
172
  }>;
148
173
  export declare const CodingRunResultSchema: z.ZodEffects<z.ZodEffects<z.ZodObject<{
149
174
  schemaVersion: z.ZodLiteral<1>;
@@ -6,6 +6,14 @@ export const MAX_CODING_SUMMARY_BYTES = 8 * 1024;
6
6
  export const MAX_CODING_TESTS = 64;
7
7
  export const MAX_CODING_TEST_COMMAND_BYTES = 2 * 1024;
8
8
  export const MAX_TAG_BYTES = 32;
9
+ /**
10
+ * A worker's output-schema failure, reduced to where and how it failed, never
11
+ * what the model wrote: `<path>:<zod issue code>`, where the path is dot-joined
12
+ * schema keys and array indices (`$` for the root). Unrecognized keys the model
13
+ * invented are carried only as the `unrecognized_keys` code, never by name.
14
+ */
15
+ export const MAX_CODING_OUTPUT_ISSUES = 8;
16
+ export const SAFE_CODING_OUTPUT_ISSUE = /^(?:\$|[a-z][A-Za-z]{0,31}(?:\.(?:[a-z][A-Za-z]{0,31}|\d{1,3})){0,7}):[a-z_]{1,32}$/;
9
17
  const MAX_REPOSITORY_INPUT_BYTES = 512;
10
18
  const MAX_REF_BYTES = 255;
11
19
  const MAX_MODEL_BYTES = 128;
@@ -108,6 +116,24 @@ const repositorySchema = z
108
116
  .transform(normalizeGitHubRepository);
109
117
  export const CodingBaseRefSchema = z.string().refine(isGitRef, "must be a safe branch ref").transform(normalizeGitRef);
110
118
  export const CodingTaskOverrideSchema = boundedText(MAX_CODING_TASK_BYTES);
119
+ /**
120
+ * A coding worker receives only its task text, so a coding agent's own
121
+ * instructions (its systemPrompt) travel inside that text, ahead of the
122
+ * request. Blank instructions leave the task unchanged. The combination must
123
+ * still fit MAX_CODING_TASK_BYTES; exceeding it is an error naming both parts
124
+ * rather than a silent truncation of either.
125
+ */
126
+ export function composeCodingTask(instructions, task) {
127
+ const standing = instructions?.trim();
128
+ if (!standing)
129
+ return task;
130
+ const composed = `Standing instructions for this coding agent:\n${standing}\n\nRequest:\n${task}`;
131
+ if (byteLength(composed) > MAX_CODING_TASK_BYTES) {
132
+ throw new Error(`The coding agent's instructions (${byteLength(standing)} bytes) plus this task (${byteLength(task)} bytes) ` +
133
+ `exceed the ${MAX_CODING_TASK_BYTES}-byte coding task limit; shorten one of them.`);
134
+ }
135
+ return composed;
136
+ }
111
137
  const runIdSchema = boundedText(MAX_RUN_ID_BYTES, true).refine((value) => /^[A-Za-z0-9][A-Za-z0-9_-]*$/.test(value), "must be an opaque identifier");
112
138
  const modelSchema = boundedText(MAX_MODEL_BYTES, true).refine((value) => /^[A-Za-z0-9][A-Za-z0-9._:-]*$/.test(value), "must be a model identifier");
113
139
  const usageSchema = z
@@ -117,6 +143,30 @@ const usageSchema = z
117
143
  costUsd: z.number().finite().nonnegative().max(MAX_COST_USD),
118
144
  })
119
145
  .strict();
146
+ const SAFE_TAG = new RegExp(`^[A-Za-z0-9][A-Za-z0-9._/-]{0,${MAX_TAG_BYTES - 1}}$`);
147
+ /**
148
+ * The tag only labels a pull request title, so a model's malformed tag must not
149
+ * sink an otherwise finished run. A tag that already passes is kept as-is; any
150
+ * other string becomes a lowercase slug (runs of disallowed characters to "-",
151
+ * no leading or trailing punctuation, at most MAX_TAG_BYTES), and one with
152
+ * nothing usable left, or a non-string, becomes null. GitHub finalization still
153
+ * re-validates whatever survives.
154
+ */
155
+ export function normalizeCodingTag(value) {
156
+ if (value === undefined || value === null)
157
+ return value;
158
+ if (typeof value !== "string")
159
+ return null;
160
+ if (SAFE_TAG.test(value))
161
+ return value;
162
+ const slug = value
163
+ .toLowerCase()
164
+ .replace(/[^a-z0-9._/-]+/g, "-")
165
+ .replace(/^[^a-z0-9]+/, "")
166
+ .slice(0, MAX_TAG_BYTES)
167
+ .replace(/[^a-z0-9]+$/, "");
168
+ return SAFE_TAG.test(slug) ? slug : null;
169
+ }
120
170
  /**
121
171
  * A short, caller-visible reference (e.g. a ticket ID) surfaced in the PR title. Never free text.
122
172
  * Accepts null as well as undefined: OpenAI's strict Structured Outputs mode requires every
@@ -126,7 +176,7 @@ const usageSchema = z
126
176
  */
127
177
  const tagSchema = z
128
178
  .string()
129
- .regex(new RegExp(`^[A-Za-z0-9][A-Za-z0-9._/-]{0,${MAX_TAG_BYTES - 1}}$`), "must be a short, safe tag")
179
+ .regex(SAFE_TAG, "must be a short, safe tag")
130
180
  .nullable()
131
181
  .optional()
132
182
  .transform((value) => value ?? undefined);
@@ -175,7 +225,8 @@ export const CodingAgentOutputSchema = z
175
225
  outcome: z.enum(["changes_ready", "no_changes", "budget_exhausted"]),
176
226
  summary: boundedText(MAX_CODING_SUMMARY_BYTES),
177
227
  tests: z.array(testResultSchema).max(MAX_CODING_TESTS),
178
- tag: tagSchema,
228
+ // Model-written, so normalized rather than rejected; see normalizeCodingTag.
229
+ tag: z.preprocess(normalizeCodingTag, tagSchema),
179
230
  })
180
231
  .strict()
181
232
  .transform((value) => ({
@@ -1,4 +1,16 @@
1
1
  import { type CodingAgentOutput } from "../coding/protocol.js";
2
+ /**
3
+ * Reads a regular file of at most `maxBytes`, refusing symlinks and anything
4
+ * that is not a regular file.
5
+ *
6
+ * Every check runs against the open handle, never against the path before
7
+ * opening it: an lstat followed by a read of the same path leaves a window in
8
+ * which the file can be swapped for a symlink or a larger file. O_NOFOLLOW
9
+ * refuses a symlink at open time, O_NONBLOCK keeps a FIFO from hanging the
10
+ * open, and fstat on the descriptor checks what was actually opened. The lstat
11
+ * afterwards covers platforms without O_NOFOLLOW: it must be the same inode.
12
+ */
13
+ export declare function readBoundedRegularFile(path: string, maxBytes: number, errorCode: string): Promise<string>;
2
14
  export declare function readCodingInput(path: string): Promise<{
3
15
  repository: string;
4
16
  schemaVersion: 1;
@@ -1,12 +1,54 @@
1
- import { lstat, mkdir, open, readFile, rename, unlink } from "node:fs/promises";
1
+ import { constants } from "node:fs";
2
+ import { lstat, mkdir, open, rename, unlink } from "node:fs/promises";
2
3
  import { dirname, join } from "node:path";
3
4
  import { MAX_CODING_ARTIFACT_BYTES, parseCodingTaskInputJson } from "../coding/protocol.js";
4
- export async function readCodingInput(path) {
5
- const metadata = await lstat(path);
6
- if (!metadata.isFile() || metadata.isSymbolicLink() || metadata.size > MAX_CODING_ARTIFACT_BYTES) {
7
- throw new Error("coding_input_invalid_file");
5
+ /**
6
+ * Reads a regular file of at most `maxBytes`, refusing symlinks and anything
7
+ * that is not a regular file.
8
+ *
9
+ * Every check runs against the open handle, never against the path before
10
+ * opening it: an lstat followed by a read of the same path leaves a window in
11
+ * which the file can be swapped for a symlink or a larger file. O_NOFOLLOW
12
+ * refuses a symlink at open time, O_NONBLOCK keeps a FIFO from hanging the
13
+ * open, and fstat on the descriptor checks what was actually opened. The lstat
14
+ * afterwards covers platforms without O_NOFOLLOW: it must be the same inode.
15
+ */
16
+ export async function readBoundedRegularFile(path, maxBytes, errorCode) {
17
+ let file;
18
+ try {
19
+ file = await open(path, constants.O_RDONLY | (constants.O_NOFOLLOW ?? 0) | (constants.O_NONBLOCK ?? 0));
20
+ }
21
+ catch (error) {
22
+ // ELOOP: the path is a symlink.
23
+ if (error.code === "ELOOP")
24
+ throw new Error(errorCode, { cause: error });
25
+ throw error;
26
+ }
27
+ try {
28
+ const metadata = await file.stat();
29
+ if (!metadata.isFile() || metadata.size > maxBytes)
30
+ throw new Error(errorCode);
31
+ const link = await lstat(path);
32
+ if (link.isSymbolicLink() || link.ino !== metadata.ino || link.dev !== metadata.dev)
33
+ throw new Error(errorCode);
34
+ const buffer = Buffer.alloc(maxBytes + 1);
35
+ let length = 0;
36
+ while (length < buffer.length) {
37
+ const { bytesRead } = await file.read(buffer, length, buffer.length - length, length);
38
+ if (bytesRead === 0)
39
+ break;
40
+ length += bytesRead;
41
+ }
42
+ if (length > maxBytes)
43
+ throw new Error(errorCode);
44
+ return buffer.subarray(0, length).toString("utf8");
45
+ }
46
+ finally {
47
+ await file.close();
8
48
  }
9
- return parseCodingTaskInputJson(await readFile(path, "utf8"));
49
+ }
50
+ export async function readCodingInput(path) {
51
+ return parseCodingTaskInputJson(await readBoundedRegularFile(path, MAX_CODING_ARTIFACT_BYTES, "coding_input_invalid_file"));
10
52
  }
11
53
  export async function writeCodingOutputAtomic(path, output) {
12
54
  const directory = dirname(path);
@@ -20,6 +20,7 @@ export declare const CODING_OUTPUT_JSON_SCHEMA: {
20
20
  };
21
21
  readonly tag: {
22
22
  readonly type: readonly ["string", "null"];
23
+ readonly description: "Optional short label shown in the pull request title, such as a ticket id. At most 32 characters: letters, digits, '.', '_', '/', '-', starting with a letter or digit, with no spaces. Example: \"add-jokes\". Use null when there is no natural label.";
23
24
  };
24
25
  readonly tests: {
25
26
  readonly type: "array";
@@ -20,7 +20,10 @@ export const CODING_OUTPUT_JSON_SCHEMA = {
20
20
  runId: { type: "string" },
21
21
  outcome: { type: "string", enum: ["changes_ready", "no_changes", "budget_exhausted"] },
22
22
  summary: { type: "string" },
23
- tag: { type: ["string", "null"] },
23
+ tag: {
24
+ type: ["string", "null"],
25
+ description: "Optional short label shown in the pull request title, such as a ticket id. At most 32 characters: letters, digits, '.', '_', '/', '-', starting with a letter or digit, with no spaces. Example: \"add-jokes\". Use null when there is no natural label.",
26
+ },
24
27
  tests: {
25
28
  type: "array",
26
29
  maxItems: 64,
@@ -1 +1,8 @@
1
+ /**
2
+ * For `coding_output_invalid`, where the model's final answer failed the output
3
+ * schema: the failing schema paths and issue codes, so the operator log can say
4
+ * which field was wrong without ever carrying the value. Undefined when the
5
+ * cause is not a schema failure.
6
+ */
7
+ export declare function safeOutputIssues(error: unknown): string[] | undefined;
1
8
  export declare function safeWorkerErrorCode(error: unknown): string;
@@ -1,3 +1,5 @@
1
+ import { ZodError } from "zod";
2
+ import { MAX_CODING_OUTPUT_ISSUES, SAFE_CODING_OUTPUT_ISSUE } from "../coding/protocol.js";
1
3
  const SAFE_WORKER_ERROR_CODES = new Set([
2
4
  "wardby_proxy_url_missing",
3
5
  "wardby_run_capability_missing",
@@ -13,6 +15,22 @@ const SAFE_WORKER_ERROR_CODES = new Set([
13
15
  "coding_output_invalid",
14
16
  "coding_output_run_mismatch",
15
17
  ]);
18
+ /**
19
+ * For `coding_output_invalid`, where the model's final answer failed the output
20
+ * schema: the failing schema paths and issue codes, so the operator log can say
21
+ * which field was wrong without ever carrying the value. Undefined when the
22
+ * cause is not a schema failure.
23
+ */
24
+ export function safeOutputIssues(error) {
25
+ const cause = error instanceof Error ? error.cause : undefined;
26
+ if (!(cause instanceof ZodError))
27
+ return undefined;
28
+ const issues = cause.issues
29
+ .map((issue) => `${issue.path.length === 0 ? "$" : issue.path.join(".")}:${issue.code}`)
30
+ .filter((issue) => SAFE_CODING_OUTPUT_ISSUE.test(issue))
31
+ .slice(0, MAX_CODING_OUTPUT_ISSUES);
32
+ return issues.length > 0 ? issues : undefined;
33
+ }
16
34
  export function safeWorkerErrorCode(error) {
17
35
  if (!(error instanceof Error))
18
36
  return "worker_failed";