@cyanheads/mcp-ts-core 0.13.8 → 0.13.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (180) hide show
  1. package/AGENTS.md +51 -24
  2. package/CLAUDE.md +51 -24
  3. package/README.md +10 -10
  4. package/changelog/0.13.x/0.13.10.md +118 -0
  5. package/changelog/0.13.x/0.13.9.md +113 -0
  6. package/dist/config/index.d.ts +3 -0
  7. package/dist/config/index.d.ts.map +1 -1
  8. package/dist/config/index.js +31 -9
  9. package/dist/config/index.js.map +1 -1
  10. package/dist/core/app.d.ts +6 -3
  11. package/dist/core/app.d.ts.map +1 -1
  12. package/dist/core/app.js +21 -5
  13. package/dist/core/app.js.map +1 -1
  14. package/dist/core/context.d.ts +114 -21
  15. package/dist/core/context.d.ts.map +1 -1
  16. package/dist/core/context.js +40 -0
  17. package/dist/core/context.js.map +1 -1
  18. package/dist/core/index.d.ts +1 -1
  19. package/dist/core/index.d.ts.map +1 -1
  20. package/dist/core/index.js.map +1 -1
  21. package/dist/core/serverManifest.d.ts +6 -0
  22. package/dist/core/serverManifest.d.ts.map +1 -1
  23. package/dist/core/serverManifest.js +6 -0
  24. package/dist/core/serverManifest.js.map +1 -1
  25. package/dist/core/worker.d.ts +6 -0
  26. package/dist/core/worker.d.ts.map +1 -1
  27. package/dist/core/worker.js +1 -0
  28. package/dist/core/worker.js.map +1 -1
  29. package/dist/linter/rules/error-contract-rules.d.ts +3 -44
  30. package/dist/linter/rules/error-contract-rules.d.ts.map +1 -1
  31. package/dist/linter/rules/error-contract-rules.js +8 -144
  32. package/dist/linter/rules/error-contract-rules.js.map +1 -1
  33. package/dist/linter/rules/index.d.ts +1 -1
  34. package/dist/linter/rules/index.d.ts.map +1 -1
  35. package/dist/linter/rules/index.js +1 -1
  36. package/dist/linter/rules/index.js.map +1 -1
  37. package/dist/linter/rules/resource-rules.d.ts.map +1 -1
  38. package/dist/linter/rules/resource-rules.js +1 -2
  39. package/dist/linter/rules/resource-rules.js.map +1 -1
  40. package/dist/linter/rules/tool-rules.d.ts +2 -1
  41. package/dist/linter/rules/tool-rules.d.ts.map +1 -1
  42. package/dist/linter/rules/tool-rules.js +37 -3
  43. package/dist/linter/rules/tool-rules.js.map +1 -1
  44. package/dist/mcp-server/handlerContext.d.ts +26 -13
  45. package/dist/mcp-server/handlerContext.d.ts.map +1 -1
  46. package/dist/mcp-server/handlerContext.js +32 -17
  47. package/dist/mcp-server/handlerContext.js.map +1 -1
  48. package/dist/mcp-server/inputRequired.d.ts +133 -12
  49. package/dist/mcp-server/inputRequired.d.ts.map +1 -1
  50. package/dist/mcp-server/inputRequired.js +192 -20
  51. package/dist/mcp-server/inputRequired.js.map +1 -1
  52. package/dist/mcp-server/prompts/prompt-registration.d.ts +10 -2
  53. package/dist/mcp-server/prompts/prompt-registration.d.ts.map +1 -1
  54. package/dist/mcp-server/prompts/prompt-registration.js +49 -11
  55. package/dist/mcp-server/prompts/prompt-registration.js.map +1 -1
  56. package/dist/mcp-server/resources/resource-registration.d.ts +4 -2
  57. package/dist/mcp-server/resources/resource-registration.d.ts.map +1 -1
  58. package/dist/mcp-server/resources/resource-registration.js +6 -4
  59. package/dist/mcp-server/resources/resource-registration.js.map +1 -1
  60. package/dist/mcp-server/resources/utils/resourceHandlerFactory.d.ts +4 -3
  61. package/dist/mcp-server/resources/utils/resourceHandlerFactory.d.ts.map +1 -1
  62. package/dist/mcp-server/resources/utils/resourceHandlerFactory.js +42 -14
  63. package/dist/mcp-server/resources/utils/resourceHandlerFactory.js.map +1 -1
  64. package/dist/mcp-server/server.d.ts +9 -0
  65. package/dist/mcp-server/server.d.ts.map +1 -1
  66. package/dist/mcp-server/server.js +14 -13
  67. package/dist/mcp-server/server.js.map +1 -1
  68. package/dist/mcp-server/tools/tool-registration.d.ts +7 -3
  69. package/dist/mcp-server/tools/tool-registration.d.ts.map +1 -1
  70. package/dist/mcp-server/tools/tool-registration.js +9 -5
  71. package/dist/mcp-server/tools/tool-registration.js.map +1 -1
  72. package/dist/mcp-server/tools/utils/inputPrevalidation.d.ts +163 -40
  73. package/dist/mcp-server/tools/utils/inputPrevalidation.d.ts.map +1 -1
  74. package/dist/mcp-server/tools/utils/inputPrevalidation.js +330 -114
  75. package/dist/mcp-server/tools/utils/inputPrevalidation.js.map +1 -1
  76. package/dist/mcp-server/tools/utils/toolHandlerFactory.d.ts +40 -19
  77. package/dist/mcp-server/tools/utils/toolHandlerFactory.d.ts.map +1 -1
  78. package/dist/mcp-server/tools/utils/toolHandlerFactory.js +387 -130
  79. package/dist/mcp-server/tools/utils/toolHandlerFactory.js.map +1 -1
  80. package/dist/mcp-server/transports/http/httpErrorHandler.d.ts.map +1 -1
  81. package/dist/mcp-server/transports/http/httpErrorHandler.js +7 -1
  82. package/dist/mcp-server/transports/http/httpErrorHandler.js.map +1 -1
  83. package/dist/mcp-server/transports/http/httpTransport.d.ts.map +1 -1
  84. package/dist/mcp-server/transports/http/httpTransport.js +65 -9
  85. package/dist/mcp-server/transports/http/httpTransport.js.map +1 -1
  86. package/dist/mcp-server/transports/stdio/stdioTransport.d.ts +9 -5
  87. package/dist/mcp-server/transports/stdio/stdioTransport.d.ts.map +1 -1
  88. package/dist/mcp-server/transports/stdio/stdioTransport.js +9 -5
  89. package/dist/mcp-server/transports/stdio/stdioTransport.js.map +1 -1
  90. package/dist/services/canvas/core/CanvasRegistry.d.ts +6 -2
  91. package/dist/services/canvas/core/CanvasRegistry.d.ts.map +1 -1
  92. package/dist/services/canvas/core/CanvasRegistry.js +7 -3
  93. package/dist/services/canvas/core/CanvasRegistry.js.map +1 -1
  94. package/dist/services/canvas/providers/duckdb/DuckdbProvider.d.ts +82 -18
  95. package/dist/services/canvas/providers/duckdb/DuckdbProvider.d.ts.map +1 -1
  96. package/dist/services/canvas/providers/duckdb/DuckdbProvider.js +620 -328
  97. package/dist/services/canvas/providers/duckdb/DuckdbProvider.js.map +1 -1
  98. package/dist/services/canvas/providers/duckdb/exportWriter.d.ts +11 -7
  99. package/dist/services/canvas/providers/duckdb/exportWriter.d.ts.map +1 -1
  100. package/dist/services/canvas/providers/duckdb/exportWriter.js +19 -16
  101. package/dist/services/canvas/providers/duckdb/exportWriter.js.map +1 -1
  102. package/dist/services/mirror/core/defineMirror.d.ts +1 -0
  103. package/dist/services/mirror/core/defineMirror.d.ts.map +1 -1
  104. package/dist/services/mirror/core/defineMirror.js +1 -0
  105. package/dist/services/mirror/core/defineMirror.js.map +1 -1
  106. package/dist/testing/index.d.ts +17 -2
  107. package/dist/testing/index.d.ts.map +1 -1
  108. package/dist/testing/index.js +21 -7
  109. package/dist/testing/index.js.map +1 -1
  110. package/dist/types-global/errors.d.ts +18 -15
  111. package/dist/types-global/errors.d.ts.map +1 -1
  112. package/dist/utils/index.d.ts +1 -1
  113. package/dist/utils/index.d.ts.map +1 -1
  114. package/dist/utils/index.js.map +1 -1
  115. package/dist/utils/internal/error-handler/errorHandler.d.ts +5 -4
  116. package/dist/utils/internal/error-handler/errorHandler.d.ts.map +1 -1
  117. package/dist/utils/internal/error-handler/errorHandler.js +9 -7
  118. package/dist/utils/internal/error-handler/errorHandler.js.map +1 -1
  119. package/dist/utils/internal/error-handler/types.d.ts +3 -1
  120. package/dist/utils/internal/error-handler/types.d.ts.map +1 -1
  121. package/dist/utils/internal/performance.d.ts +4 -2
  122. package/dist/utils/internal/performance.d.ts.map +1 -1
  123. package/dist/utils/internal/performance.js +8 -6
  124. package/dist/utils/internal/performance.js.map +1 -1
  125. package/dist/utils/internal/telemetryMessages.d.ts +0 -1
  126. package/dist/utils/internal/telemetryMessages.d.ts.map +1 -1
  127. package/dist/utils/internal/telemetryMessages.js +0 -1
  128. package/dist/utils/internal/telemetryMessages.js.map +1 -1
  129. package/dist/utils/network/pacer.d.ts +38 -5
  130. package/dist/utils/network/pacer.d.ts.map +1 -1
  131. package/dist/utils/network/pacer.js +87 -25
  132. package/dist/utils/network/pacer.js.map +1 -1
  133. package/dist/utils/telemetry/attributes.d.ts +10 -5
  134. package/dist/utils/telemetry/attributes.d.ts.map +1 -1
  135. package/dist/utils/telemetry/attributes.js +10 -5
  136. package/dist/utils/telemetry/attributes.js.map +1 -1
  137. package/framework-skills/add-app-tool/SKILL.md +3 -3
  138. package/framework-skills/add-export/SKILL.md +5 -16
  139. package/framework-skills/add-prompt/SKILL.md +7 -3
  140. package/framework-skills/add-resource/SKILL.md +7 -5
  141. package/framework-skills/add-service/SKILL.md +3 -12
  142. package/framework-skills/add-test/SKILL.md +6 -3
  143. package/framework-skills/add-tool/SKILL.md +40 -42
  144. package/framework-skills/api-auth/SKILL.md +2 -2
  145. package/framework-skills/api-canvas/SKILL.md +17 -8
  146. package/framework-skills/api-config/SKILL.md +5 -4
  147. package/framework-skills/api-context/SKILL.md +168 -42
  148. package/framework-skills/api-errors/SKILL.md +48 -51
  149. package/framework-skills/api-linter/SKILL.md +30 -35
  150. package/framework-skills/api-mirror/SKILL.md +2 -1
  151. package/framework-skills/api-telemetry/SKILL.md +14 -10
  152. package/framework-skills/api-testing/SKILL.md +43 -11
  153. package/framework-skills/api-utils/SKILL.md +2 -2
  154. package/framework-skills/api-workers/SKILL.md +3 -1
  155. package/framework-skills/design-mcp-server/SKILL.md +6 -6
  156. package/framework-skills/field-test/SKILL.md +5 -5
  157. package/framework-skills/git-wrapup/SKILL.md +8 -6
  158. package/framework-skills/orchestrations/SKILL.md +7 -6
  159. package/framework-skills/orchestrations/workflows/field-test-fix.md +9 -19
  160. package/framework-skills/orchestrations/workflows/fix-wrapup-release.md +7 -7
  161. package/framework-skills/orchestrations/workflows/greenfield-build.md +8 -5
  162. package/framework-skills/orchestrations/workflows/maintenance-release.md +8 -8
  163. package/framework-skills/polish-docs-meta/SKILL.md +4 -4
  164. package/framework-skills/release-and-publish/SKILL.md +8 -6
  165. package/framework-skills/release-pr-review/SKILL.md +38 -24
  166. package/framework-skills/report-issue-framework/SKILL.md +7 -5
  167. package/framework-skills/report-issue-local/SKILL.md +8 -6
  168. package/framework-skills/security-pass/SKILL.md +14 -13
  169. package/package.json +6 -5
  170. package/scripts/devcheck.ts +7 -6
  171. package/scripts/install-otel.ts +84 -0
  172. package/scripts/lint-mcp.ts +87 -27
  173. package/scripts/lint-packaging.ts +226 -4
  174. package/scripts/release-github.ts +117 -5
  175. package/templates/.env.example +2 -0
  176. package/templates/AGENTS.md +5 -4
  177. package/templates/CLAUDE.md +5 -4
  178. package/templates/Dockerfile +67 -50
  179. package/templates/_.mcpbignore +2 -0
  180. package/templates/src/mcp-server/tools/definitions/echo.tool.ts +3 -8
@@ -4,7 +4,7 @@ description: >
4
4
  Testing patterns for MCP tool/resource handlers using `createMockContext` and Vitest. Covers mock context options, handler testing, McpError assertions, format testing, Vitest config setup, and test isolation conventions.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "1.12"
7
+ version: "1.13"
8
8
  audience: external
9
9
  type: reference
10
10
  ---
@@ -133,6 +133,8 @@ toolContractSuite(searchTool, {
133
133
 
134
134
  Use `runToolContract(definition, input, { context })` from `/testing` when a custom test runner or an imperative assertion is a better fit. It intentionally skips transport auth and telemetry; those belong in transport/integration tests.
135
135
 
136
+ A declared reason thrown without a hint — a bare `ctx.fail('reason')` or a service throw carrying `{ reason }` — comes back with the entry's `recovery` as `data.recovery.hint` and a `Recovery:` line in `content[]`, as in production. The one production field it leaves out is `data.requestId` (and the `request <id>` term closing `content[]`), since there is no real request; a test asserting the factory's envelope instead expects both. Calling `definition.handler(...)` directly returns the `McpError` exactly as the throw site built it — no fill, no request id.
137
+
136
138
  Arguments that fail the `input` schema are rejected the way the production handler factory rejects them: `InvalidParams` (`-32602`), with a message naming the tool and every failing field. That is the code a client sees on the wire, so assert it — not `ValidationError` (`-32007`), which stays the classification for a `ZodError` a handler throws itself. A result that breaks the tool's own `output` or `enrichment` schema is the definition's bug, so it returns `InternalError` (`-32603`) with a message naming that contract, exactly as in production.
137
139
 
138
140
  Cancellation settles as it does in production. Pass `context: { signal }` and abort it: once the signal has fired, whatever the handler — or the output validation, `format()`, and enrichment after it — throws comes back as `RequestCancelled` (`-32011`), whether that is the signal's `AbortError`, its reason string, a `withRetry` backoff that stopped, or an `McpError` of the handler's own. A throw while the signal is still live keeps its own classification, and argument parsing stays outside the settle, so schema-invalid arguments on an aborted signal still return `InvalidParams`. A `toolContractSuite` error case with an aborted `context.signal` asserts `code: JsonRpcErrorCode.RequestCancelled` the same way.
@@ -149,6 +151,7 @@ createMockContext({ tenantId: 'test-tenant' }) // explicit tenant
149
151
  createMockContext({ errors: myTool.errors }) // attaches typed ctx.fail keyed by the contract reasons
150
152
  createMockContext({ inputResponses: { confirm: { action: 'accept', content: { ok: true } } } }) // second round of a multi-round-trip handler
151
153
  createMockContext({ requestState: 'opaque-state' }) // seeds ctx.inputs.state()
154
+ createMockContext({ clientCapabilities: { roots: {} } }) // seeds ctx.clientCapabilities and filters inputResponses to declared kinds
152
155
  createMockContext({ requestId: 'my-id' }) // override request ID (default: 'test-request-id')
153
156
  createMockContext({ notifyResourceListChanged: () => {} }) // with resource-list change notifier
154
157
  createMockContext({ notifyResourceUpdated: (_uri) => {} }) // with resource update notifier
@@ -162,6 +165,7 @@ createMockContext({ uri: new URL('myscheme://item/123') }) // for resource han
162
165
  ```ts
163
166
  interface MockContextOptions<TErrors extends readonly ErrorContract[] | undefined> {
164
167
  auth?: AuthContext;
168
+ clientCapabilities?: ClientCapabilities;
165
169
  errors?: TErrors | undefined;
166
170
  inputResponses?: InputResponses | Record<string, unknown>;
167
171
  notifyPromptListChanged?: () => void;
@@ -181,6 +185,7 @@ interface MockContextOptions<TErrors extends readonly ErrorContract[] | undefine
181
185
  |:-------|:-------|
182
186
  | _(none)_ | Working `ctx.state` on tenant `'default'`; `ctx.inputs` is empty (first round) |
183
187
  | `auth` | Sets `ctx.auth` for scope-checking tests |
188
+ | `clientCapabilities` | Sets `ctx.clientCapabilities` (`undefined` when omitted) and applies the production filter to `inputResponses`: only the answers these capabilities cover reach `ctx.inputs` (elicit → `elicitation`, and `elicitation.form` when it carries `content`; sampling → `sampling`, and `sampling.tools` when it holds a `tool_use` / `tool_result` block; roots → `roots`). Omitted, every seeded response reaches `ctx.inputs`, so existing `{ inputResponses }` tests are unaffected. The mock's `ctx.requestInput` stays ungated either way |
184
189
  | `errors` | Attaches a typed `ctx.fail` against the contract — same wiring the production handler factory uses. Pass `myTool.errors` directly; the return type narrows to `HandlerContext<ReasonOf<…>>`, so the context is assignable to that definition's handler parameter. |
185
190
  | `inputResponses` | Seeds `ctx.inputs` with the responses a retried request would carry, keyed by the identifiers the handler's `ctx.requestInput(...)` assigned (see below) |
186
191
  | `notifyPromptListChanged` | Assigns `ctx.notifyPromptListChanged` for prompt-list change notification tests |
@@ -219,14 +224,14 @@ Reach for `createInMemoryStorage()` when a service takes a `StorageService` dire
219
224
 
220
225
  ### Mock inputs
221
226
 
222
- `ctx.requestInput` is the real implementation: it throws an `InputRequiredSignal` the production handler factories convert into an `input_required` result. In a unit test the handler is called directly, so that signal surfaces as a thrown value — which is exactly how you assert the first round.
227
+ `ctx.requestInput` is the real implementation: it throws an `InputRequiredSignal` the production handler factories convert into an `input_required` result. In a unit test the handler is called directly, so that signal surfaces as a thrown value — which is exactly how you assert the first round. The examples below drive `export_report` from `api-context` § *The shape of a multi-round-trip handler*, which asks for a format the caller left out:
223
228
 
224
229
  ```ts
225
230
  import { isInputRequiredSignal } from '@cyanheads/mcp-ts-core';
226
231
 
227
- it('asks for confirmation on the first round', async () => {
232
+ it('asks for the format on the first round', async () => {
228
233
  const ctx = createMockContext();
229
- await expect(myTool.handler(myTool.input.parse({ path: '/tmp/x' }), ctx))
234
+ await expect(exportReport.handler(exportReport.input.parse({ reportId: 'r1' }), ctx))
230
235
  .rejects.toSatisfy(isInputRequiredSignal);
231
236
  });
232
237
  ```
@@ -236,30 +241,57 @@ To assert on *what* was requested, use `expectInputRequired` from `/testing`. It
236
241
  ```ts
237
242
  import { createMockContext, expectInputRequired } from '@cyanheads/mcp-ts-core/testing';
238
243
 
239
- const asked = await expectInputRequired(() => myTool.handler(input, createMockContext()));
240
- expect(asked.inputRequests?.confirm?.method).toBe('elicitation/create');
244
+ const asked = await expectInputRequired(() => exportReport.handler(input, createMockContext()));
245
+ expect(asked.inputRequests?.format?.method).toBe('elicitation/create');
241
246
  ```
242
247
 
243
248
  Pass `asked.requestState` back as `createMockContext({ requestState })` when the handler reads state from the prior round. `inputResponses` drives the second round. `ctx.inputs.accepted(key, schema)` and `.view(key)` read it with the same helpers production uses, so a wrong response shape fails in the test:
244
249
 
245
250
  ```ts
246
- it('proceeds once the user accepts', async () => {
251
+ it('exports in the format the user picked', async () => {
247
252
  const ctx = createMockContext({
248
- inputResponses: { confirm: { action: 'accept', content: { confirm: true } } },
253
+ inputResponses: { format: { action: 'accept', content: { format: 'csv' } } },
249
254
  });
250
- await expect(myTool.handler(input, ctx)).resolves.toMatchObject({ deleted: '/tmp/x' });
255
+ await expect(exportReport.handler(input, ctx)).resolves.toMatchObject({ url: expect.any(String) });
251
256
  });
252
257
 
253
258
  it('stops when the user declines', async () => {
254
259
  const ctx = createMockContext({
255
- inputResponses: { confirm: { action: 'decline' } },
260
+ inputResponses: { format: { action: 'decline' } },
256
261
  });
257
- await expect(myTool.handler(input, ctx)).rejects.toThrow(McpError);
262
+ await expect(exportReport.handler(input, ctx)).rejects.toThrow(McpError);
258
263
  });
259
264
  ```
260
265
 
266
+ Seeding an answer this way is right for a handler that treats it as input. A consent gate does not: it acts only on a record it stored when it asked, so an answer seeded alone makes it ask again — see below.
267
+
261
268
  `ctx.inputs.dropped` is always `[]` on a mock context — the drop only happens in the SDK's wire decoding, so cover it in an integration test rather than a unit one.
262
269
 
270
+ Seed `clientCapabilities` to test what a client without a capability gets: a pre-answered `inputResponses` the declared capabilities do not cover — a kind never declared, a form answer (one carrying `content`) from a client that declared only `elicitation.url`, a tool-use sampling answer without `sampling.tools` — never reaches `ctx.inputs`, so the handler asks again. A handler that falls through when a capability is missing (`if (ctx.clientCapabilities?.roots) … else …`) is tested by seeding both shapes.
271
+
272
+ A consent gate that redeems a `ctx.state` record (see `api-context` § *Consent gates*) needs the record in the second round's storage, and each mock context has its own. Copy what round one stored into the round-two context. The record carries the caller, so seed the same `auth` (or none) on both:
273
+
274
+ ```ts
275
+ const accept = { confirm: { action: 'accept', content: { confirm: true } } };
276
+
277
+ const first = createMockContext();
278
+ const asked = await expectInputRequired(() => deletePath.handler(input, first));
279
+ const record = await first.state.get(`consent/${asked.requestState}`);
280
+
281
+ const second = createMockContext({ inputResponses: accept, requestState: asked.requestState });
282
+ await second.state.set(`consent/${asked.requestState}`, record);
283
+ await expect(deletePath.handler(input, second)).resolves.toEqual({ deleted: input.path });
284
+
285
+ // The record is spent: the same state again asks for a fresh confirmation.
286
+ await expect(deletePath.handler(input, second)).rejects.toSatisfy(isInputRequiredSignal);
287
+
288
+ // An answer with no record behind it asks as well — nothing was asked, so nothing was confirmed.
289
+ const unasked = createMockContext({ inputResponses: accept });
290
+ await expect(deletePath.handler(input, unasked)).rejects.toSatisfy(isInputRequiredSignal);
291
+ ```
292
+
293
+ A record the round-two context holds under another `auth`, or one written by another tool (its `operation` differs), asks again the same way — cover whichever of those the handler's binding is meant to catch.
294
+
263
295
  ### Mock logger
264
296
 
265
297
  `ctx.log` captures all log calls for inspection. Import `MockContextLogger` from `@cyanheads/mcp-ts-core/testing` and cast `ctx.log` to access the `.calls` array (the cast is necessary because `createMockContext` returns `Context`, which types `log` as `ContextLogger`):
@@ -4,7 +4,7 @@ description: >
4
4
  API reference for all utilities exported from `@cyanheads/mcp-ts-core/utils`. Use when looking up utility method signatures, options, peer dependencies, or usage patterns.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "2.13"
7
+ version: "2.14"
8
8
  audience: external
9
9
  type: reference
10
10
  ---
@@ -37,7 +37,7 @@ Utility exports from `@cyanheads/mcp-ts-core/utils`. Utilities with complex APIs
37
37
  | `deadlineMs` | `RetryOptions` field | One wall-clock budget across every attempt, backoff, and honored `Retry-After` — the bound `maxRetries` plus a per-attempt timeout cannot express. Four 30s attempts outlast a client's 60s request timeout, so the caller gets a transport timeout instead of the server's classified error. **Thread `attempt.signal` into the attempt's I/O** (`fetchWithTimeout(url, Math.min(30_000, remainingMs), ctx, { signal })`) or the deadline overshoots by one in-flight request. Clock is `AbortController` + `setTimeout` (never `AbortSignal.timeout()`, per the Bun realm mismatch), cleared on return — no timer outlives the call. Expiry rejects with `Timeout` (-32004) carrying `data: { reason: 'retry_deadline_exceeded', deadlineMs, elapsedMs, retryAttempts }` and the last attempt's error as `cause`; **one shape for every expiry**, including the per-attempt `Timeout` (`errorSource: 'FetchSignalTimeout'`) the clock's abort raises inside `fetchWithTimeout` and the raw abort reason a mid-backoff expiry would otherwise surface. No `retryable` flag (a narrower call can still succeed) and no `attempt` index (`retryAttempts` carries it). A backoff that would outlast the remaining budget fails fast with the expiry instead of sleeping into a certain timeout; an honored `Retry-After` that would outlast it takes the `maxDelayMs` exit instead — the attempt's error unchanged, `data.retryAfter` intact, since "wait the window the upstream named" is still the caller's action. **Three clocks stay distinct:** a caller abort on `options.signal` keeps precedence — mid-attempt it rethrows the attempt's error unchanged, mid-backoff it rejects with `signal.reason` itself (an `AbortError` `DOMException` for a reason-less `abort()`), and the handler factory reports either as `RequestCancelled` when the request signal is the one that fired — a single attempt's timeout is `Timeout` with `errorSource: 'FetchTimeout'` and no `reason`, and the expiry is `Timeout` with the `reason`. Unset, behavior is identical to before — attempt counts, delays, log lines, and the exhausted-error shape untouched. Bounds **one** ladder: a tool making three upstream calls threads its own remaining budget into each. |
38
38
  | `defaultIsTransient` | `(error: unknown) -> boolean` | The predicate `withRetry` uses when `isTransient` is omitted: an `McpError` with a transient code (`ServiceUnavailable`, `Timeout`, `RateLimited`) unless it carries `data.retryable === false`, `data.reason === 'pacer_shed'`, or `data.errorSource === 'FetchSignalTimeout'` (a caller-side deadline that already fired); any non-`McpError` throw is assumed transient. Exported so `isTransient` — which **replaces** the default outright — can compose instead of mirroring the transient set, which drifts silently when the framework's classification changes: `isTransient: (error) => !isMyBudgetRefusal(error) && defaultIsTransient(error)`, or the inverse `defaultIsTransient(error) \|\| isMyRetryableShape(error)`. The transient code set itself stays private (a module-level `Set` an exported binding could be mutated into framework-wide retry behavior). |
39
39
  | `httpErrorFromResponse` | `(response: Response, options?: HttpErrorFromResponseOptions) -> Promise<McpError>` | Maps an HTTP `Response` to a properly classified `McpError` — full status table including 401/403/408/422/429/5xx, body capture (truncated), `retry-after` header, optional `cause`. `error.data` carries `status`/`body` plus the legacy `statusCode`/`responseBody` aliases (identical values), so a consumer can classify either helper's error without knowing which raised it. Use this instead of hand-rolling `if (status === 429) ...` ladders. Reads the response body — `clone()` first if you need it elsewhere. **`error.data` is client-facing** — the framework forwards it verbatim as `structuredContent.error.data` — so the full upstream URL is **omitted by default**: a request URL routinely carries user input, internal identifiers, or an API key in its query string. `includeUrl: true` opts into `data.url` carrying the full `response.url`; with an empty `response.url` no key is added either way, and the message still names the host. Response headers are opt-in on the same footing: `errorHeaders: ['x-ratelimit-remaining-usd', 'x-request-id']` copies the named headers onto `data.headers` under **lowercase** keys — selection is case-insensitive and entries differing only in case collapse to one key, presence follows `Headers.has()` (an empty value is captured as `''`, an absent header adds no key), and a multi-valued field is captured comma-joined as `Headers.get()` returns it. Omitted, empty, or matching nothing, no `headers` key is emitted. `set-cookie` is **never** captured whatever the selector says: it is credential-bearing and `Headers.get()` joins its values into a string that is not a valid reconstruction. Every selected value reaches the client, so never name a header that carries a credential — and a selected `Location` can itself carry a sensitive path, query, or token. `HttpErrorFromResponseOptions`: `service?` (logical name in message, e.g. `'NCBI'`), `captureBody?` (default `true`), `bodyLimit?` (default `500`), `includeUrl?` (default `false`), `errorHeaders?` (default none), `data?` (extra fields merged into `error.data`, overriding defaults on key collision — a caller's own `url` or `headers` still reaches the wire), `cause?`, `codeOverride?` (per-status mapping override). Pairs naturally with `withRetry` — both classify codes the same way. A 501 also carries `data.retryable: false`, so retry fails it fast instead of re-asking for a method the upstream does not implement. |
40
- | `createPacer` | `(options: PacerOptions) -> Pacer` | FIFO queue in front of one rate-limited upstream — the outbound counterpart to `RateLimiter` (`utils/security`), which is inbound, per-caller, and reject-only, so it cannot queue work against an upstream budget. `pacer.run(task, { signal?, maxWaitMs? })` holds `task` until every `limits` window, `minStartGapMs`, `maxConcurrent`, and the cooldown gate allow it, then calls it with the caller's signal. `PacerOptions`: `name` (author-set telemetry label), `limits` (`{ requests, perMs }[]` — each a sliding window over recorded **start** times, so a slow response never widens the rate the upstream sees; all must allow a start), `minStartGapMs` (**not** expressible through `limits`: `{ requests: 10, perMs: 1000 }` permits ten starts in the same millisecond), `maxConcurrent`, `maxQueueDepth` (absolute backpressure for callers passing no `maxWaitMs`; rejects without arming a timer), `cooldown` (`{ baseMs, maxMs }`). **Shed:** `maxWaitMs` bounds queue time only, never the task. The projected wait is exact over the windows and the gap but a lower bound once `maxConcurrent` binds (a slot frees on an unknowable completion), so enqueue rejects only when that lower bound already exceeds `maxWaitMs` — no false sheds — and a still-queued entry rejects when `maxWaitMs` elapses. The shed error is `rateLimited` (-32003) with `data: { reason: 'pacer_shed', retryAfter, queueDepth }` and **no `retryable: false`** — to the calling agent a shed is an ordinary rate limit (wait `retryAfter`, call again) and that flag would say the opposite; `defaultIsTransient` reads the `reason` instead, so an enclosing `withRetry` fails fast rather than sleeping past the deadline the shed enforces. **Cooldown gate:** a `RateLimited` thrown by the task closes the gate for every queued caller until an absolute instant, `min(max(baseMs · 2^(consecutive−1), retryAfter), maxMs)` — `maxMs` caps both the doubling and an honored `Retry-After`, so a pathological upstream value cannot park the queue. Absent or unparseable `retryAfter` leaves the doubling; any other error leaves the gate open; the first success resets the count. **Composition:** `withRetry(({ signal }) => pacer.run(fn, { signal }), { signal, deadlineMs })` — retry outside, pacer inside, so each attempt re-queues and is re-paced. Because the gate is an absolute instant rather than a duration counted from dequeue, retry's `Retry-After` sleep and the gate overlap in wall-clock instead of summing: the window is waited once, not twice. **Lifecycle:** timers and `AbortSignal` only, process-local; the dispatch timer is `unref()`'d where supported; `dispose()` / `[Symbol.dispose]()` clears it and rejects queued waiters with `RequestCancelled` (in-flight tasks are left to finish) — wire it through `createApp({ teardown })`. On Workers state is per-isolate so the limits bind per isolate, OTel is off so the metrics are inert, and `createWorkerHandler` accepts no `teardown`. Metrics: `mcp.pacer.queue_depth`, `mcp.pacer.wait`, `mcp.pacer.sheds`, `mcp.pacer.cooldowns`, attributed by `mcp.pacer.name` only — see `api-telemetry`. |
40
+ | `createPacer` | `(options: PacerOptions) -> Pacer` | FIFO queue in front of one rate-limited upstream — the outbound counterpart to `RateLimiter` (`utils/security`), which is inbound, per-caller, and reject-only, so it cannot queue work against an upstream budget. `pacer.run(task, { signal?, maxWaitMs? })` holds `task` until every `limits` window, `minStartGapMs`, `maxConcurrent`, and the cooldown gate allow it, then calls it with the caller's signal. `PacerOptions`: `name` (author-set telemetry label), `limits` (`{ requests, perMs }[]` — each a sliding window over recorded **start** times, so a slow response never widens the rate the upstream sees; all must allow a start), `minStartGapMs` (**not** expressible through `limits`: `{ requests: 10, perMs: 1000 }` permits ten starts in the same millisecond), `maxConcurrent`, `maxQueueDepth` (absolute backpressure for callers passing no `maxWaitMs`; rejects without arming a timer; bounds **waiters only** — an arrival whose slot is open that instant starts without queueing, so `0` means "run when a slot is free, never wait"), `cooldown` (`{ baseMs, maxMs }`). **Shed:** `maxWaitMs` bounds queue time only, never the task. The projected wait is exact over the windows and the gap but a lower bound once `maxConcurrent` binds (a slot frees on an unknowable completion), so enqueue rejects only when that lower bound already exceeds `maxWaitMs` — no false sheds — and a still-queued entry rejects when `maxWaitMs` elapses, unless its slot opens that same instant. The shed error is `rateLimited` (-32003) with `data: { reason: 'pacer_shed', shedKind, retryAfter, queueDepth }`. `shedKind` (`PacerShedKind`) is `queue_full` (the call would wait behind `maxQueueDepth` waiters), `wait_projected` (the enqueue projection exceeds `maxWaitMs`), or `wait_elapsed` (`maxWaitMs` ran out while queued), and the message follows the kind — a `queue_full` shed names the full queue, not a wait budget. `retryAfter` is seconds until a caller joining behind every remaining waiter could start; while `maxConcurrent` is saturated — a release the projection cannot see — it is floored at the longest wait of any queued caller, the shed one included, minimum 1. `queueDepth` is the waiters still queued. **No `retryable: false`** — to the calling agent a shed is an ordinary rate limit (wait `retryAfter`, call again) and that flag would say the opposite; `defaultIsTransient` reads the `reason` instead, so an enclosing `withRetry` fails fast rather than sleeping past the deadline the shed enforces. **Cooldown gate:** a `RateLimited` thrown by the task closes the gate for every queued caller until an absolute instant, `min(max(baseMs · 2^(consecutive−1), retryAfter), maxMs)` — `maxMs` caps both the doubling and an honored `Retry-After`, so a pathological upstream value cannot park the queue. Absent or unparseable `retryAfter` leaves the doubling; any other error leaves the gate open and the count untouched, and a shed (`reason: 'pacer_shed'`) from a pacer nested inside the task is local backpressure, never a gate closure. The first success resets the count, and so does a gate that has stood open for `maxMs`: the next rate limit starts over at `baseMs`, while one arriving sooner — the gate still closed included — keeps doubling, so continuous demand under a sustained limit keeps its capped backoff. **`pacer.cooldown`** samples the gate as `PacerCooldownState` `{ remainingMs, consecutive }`: `remainingMs` is the shared gate, not one rate limit's own computation (rate limits landing together close one gate at the later instant), so a task's rejection handler can report it on the server's own error — the pacer never writes to the task's error. Both stay 0 without `cooldown`. **Composition:** `withRetry(({ signal }) => pacer.run(fn, { signal }), { signal, deadlineMs })` — retry outside, pacer inside, so each attempt re-queues and is re-paced. Because the gate is an absolute instant rather than a duration counted from dequeue, retry's `Retry-After` sleep and the gate overlap in wall-clock instead of summing: the window is waited once, not twice. **Lifecycle:** timers and `AbortSignal` only, process-local; the dispatch timer is `unref()`'d where supported; `dispose()` / `[Symbol.dispose]()` clears it and rejects queued waiters with `RequestCancelled` (in-flight tasks are left to finish) — wire it through `createApp({ teardown })`. On Workers state is per-isolate so the limits bind per isolate, OTel is off so the metrics are inert, and `createWorkerHandler` accepts no `teardown`. Metrics: `mcp.pacer.queue_depth`, `mcp.pacer.wait`, `mcp.pacer.sheds`, `mcp.pacer.cooldowns`, attributed by `mcp.pacer.name` only — see `api-telemetry`. |
41
41
  | `httpStatusToErrorCode` | `(status: number) -> JsonRpcErrorCode \| undefined` | Sync status → code lookup. Returns `undefined` for 1xx/2xx. A 3xx maps to `InvalidRequest` — it reaches error mapping under `redirect: 'manual'`, where the request as sent cannot be served at this URL, and that code is outside `withRetry`'s transient set since re-issuing returns the same redirect. Use when you need just the code without a `Response` object handy. No status maps to `InternalError` — that code means *this* server failed, which a remote status cannot establish; every 5xx is `ServiceUnavailable` (or `Timeout` for 504) and so picks up `withRetry`'s default transient policy. |
42
42
 
43
43
  ---
@@ -4,7 +4,7 @@ description: >
4
4
  Cloudflare Workers deployment using `createWorkerHandler` from `@cyanheads/mcp-ts-core/worker`. Covers the full handler signature, binding types, CloudflareBindings extensibility, runtime compatibility guards, and wrangler.toml requirements.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "1.8"
7
+ version: "1.9"
8
8
  audience: external
9
9
  type: reference
10
10
  ---
@@ -204,6 +204,8 @@ export function getServerConfig() {
204
204
 
205
205
  **`in-memory` storage is volatile.** Data stored with the `in-memory` provider is lost between cold starts and is not shared across Worker instances. Use `cloudflare-kv`, `cloudflare-r2`, or `cloudflare-d1` for any state that must persist or be shared.
206
206
 
207
+ **Multi-round-trip retries land on any isolate.** Each 2026-07-28 request builds its own `McpServer`, and the retry after an `input_required` result can be routed to a different isolate. Set `MCP_REQUEST_STATE_KEY` (≥ 32 bytes) as a Worker secret — it is a core binding, injected like the rest — so every isolate seals and verifies `requestState` with the same key, and keep a consent gate's record (`api-context` § *Consent gates*) in `cloudflare-d1`. `in-memory` loses it to the next isolate, so the retry asks again. Never `cloudflare-kv`: it is eventually consistent, so a retry served elsewhere may not see the record yet, and a deletion may not yet stop an immediate replay — it widens the window concurrent retries already have, since redeeming is not atomic on any provider until `ctx.state` gains a `take` ([#593](https://github.com/cyanheads/mcp-ts-core/issues/593)).
208
+
207
209
  **Node-only utilities throw in Workers.** `scheduler` (`node-cron`), `sanitizePath` (fs-based), and `filesystem` storage provider all throw `ConfigurationError` when called from a Worker. Guard with `runtimeCaps.isNode` or avoid entirely.
208
210
 
209
211
  **DataCanvas is unavailable in Workers.** DuckDB has no V8-isolate build, so `core.canvas` is always `undefined` on Workers. Setting `CANVAS_PROVIDER_TYPE=duckdb` (the only non-default value) in `wrangler.toml` triggers a fail-closed `ConfigurationError` at init time:
@@ -4,7 +4,7 @@ description: >
4
4
  Design the tool surface, resources, and service layer for a new MCP server. Use when starting a new server, planning a major feature expansion, or when the user describes a domain/API they want to expose via MCP. Produces a design doc at docs/design.md that drives implementation.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "2.29"
7
+ version: "2.31"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -263,9 +263,9 @@ A reference tool is the surface's decoder ring: which codes exist, what they mea
263
263
 
264
264
  Tools that perform multi-step mutations (the Workflow shape) have two safety considerations beyond single-call tools. Both are about giving the agent — and the human behind it — a chance to catch a bad invocation before it commits.
265
265
 
266
- **Confirmation-gated destructive modes, with an annotation fallback.** When a workflow's `mode` parameter switches between safe and destructive arms (`draft` vs `send`, `plan` vs `apply`), gate the destructive arm on a confirmation the handler asks for via `ctx.requestInput(...)`, so a human approves before the irreversible step fires. The handler is re-entered with the answer on `ctx.inputs`; it does not `await` mid-call.
266
+ **Confirmation-gated destructive modes, with an annotation fallback.** When a workflow's `mode` parameter switches between safe and destructive arms (`draft` vs `send`, `plan` vs `apply`), gate the destructive arm on a confirmation the handler asks for via `ctx.requestInput(...)`, so a human approves before the irreversible step fires. The handler is re-entered with the answer on `ctx.inputs`; it does not `await` mid-call. The answer alone does not prove the prompt was shown — a client can send one unasked, and any `requestState` replays within its lifetime — so the gate stores the operation, the caller, the confirmed target, and a hash of what it holds in a `ctx.state` record keyed by a random id, sends only that id as `requestState`, and redeems the record before the destructive arm runs (`api-context` § *Consent gates*). Plan shared storage for that record when a 2026-07-28 retry can reach another instance (`filesystem`, `supabase`, or `cloudflare-d1` — not `cloudflare-kv`), and set `MCP_REQUEST_STATE_KEY` so a retry can only carry state the server minted. Redeeming is not atomic — concurrent retries on one id can each pass the gate until `ctx.state` gains an atomic `take` ([#593](https://github.com/cyanheads/mcp-ts-core/issues/593)) — so an arm that must not run twice is idempotent per record.
267
267
 
268
- The gate is always *reachable* — `ctx.requestInput` is present on every transport and both protocol revisions (2025-11-25 legacy, 2026-07-28 current) — but it is not always *answerable*: a client that never fulfils the `input_required` result simply doesn't retry, and the destructive step never runs. The same holds for a 2025-11-25 HTTP client when the server runs `MCP_SESSION_MODE=stateless`: the legacy round-trip shim still runs, but its capability gate refuses because the serving instance never processed `initialize` — the destructive step never fires. That is the safe outcome, but it makes the tool unusable for those clients, so a server built around such a gate declares `createApp({ sessionMode: { default: 'stateful', require: 'stateful' } })` and refuses to start stateless rather than degrading (`api-context` § `ctx.requestInput`). Keep `destructiveHint: true` in annotations so those clients' own approval flows still surface the risk. A decline is terminal — the handler fails the call rather than re-asking, which would loop until the round budget runs out. The handler shape is in `api-context` § *The shape of a multi-round-trip handler*.
268
+ The gate is always *reachable* — `ctx.requestInput` is present on every transport and both protocol revisions (2025-11-25 legacy, 2026-07-28 current) — but it is not always *answerable*: a client that never fulfils the `input_required` result simply doesn't retry, and — with the record pattern, which proceeds only on a record it redeemed — the destructive step never runs. The same holds for a 2025-11-25 HTTP client when the server runs `MCP_SESSION_MODE=stateless`: the legacy round-trip shim still runs, but its capability gate refuses because the serving instance never processed `initialize` — the destructive step never fires. That is the safe outcome, but it makes the tool unusable for those clients, so a server built around such a gate declares `createApp({ sessionMode: { default: 'stateful', require: 'stateful' } })` and refuses to start stateless rather than degrading (`api-context` § `ctx.requestInput`). Keep `destructiveHint: true` in annotations so those clients' own approval flows still surface the risk. A decline is terminal — the handler fails the call rather than re-asking, which would loop until the round budget runs out. `ctx.clientCapabilities` never decides whether to ask: a gate that skips its prompt when `elicitation` is undeclared is the bypass the gate exists to prevent.
269
269
 
270
270
  **Safe defaults on parameters that determine blast radius.** When a workflow accepts a parameter that controls how far-reaching a mutation is, default to the safer value. A bulk file-update tool defaulting `mode: 'preview'` (no writes) means a sloppy agent call shows a diff rather than blasting changes; an apply-plan tool defaulting `dryRun: true` means a misread plan previews rather than executes; an object-delete tool requiring an explicit `confirmCount` matching the result-set size means an unscoped query can't silently nuke a million rows. Agents that genuinely want the destructive behavior have to name it explicitly, which surfaces intent in the tool call and in logs.
271
271
 
@@ -332,7 +332,7 @@ nctIds: z.union([z.string(), z.array(z.string()).max(5)])
332
332
  | Delimiter-joined list where an array is accepted | `"US,JP,KR"` → `["US","JP","KR"]` | Split on the documented separator |
333
333
  | Spelled-out vs. abbreviated name | `"Houston, Texas"` → `"Houston, TX"` | Normalize against the bundled name table |
334
334
 
335
- These are **value**-level, and the mappings are domain knowledge — settle them per input in the design doc's param table. Argument **key** names are not: the framework drops client-added root keys and rewrites declared and case-style key aliases before the schema sees the arguments, and repairs a JSON-stringified array against the tool's own schema after a failed parse. Don't re-implement any of that per server — see `add-tool` § *Three things the framework fixes before the schema sees the arguments*.
335
+ These are **value**-level, and the mappings are domain knowledge — settle them per input in the design doc's param table. Argument **key** names are not: the framework rewrites declared and case-style key aliases and drops client-added root keys before the schema sees the arguments, and repairs a JSON-stringified array or object, or an integer sent for a string, against the tool's own schema after a failed parse — so an ID field stays `z.string()`, never a `string | number` union. Don't re-implement any of that per server — see `add-tool` § *Three things the framework fixes before the schema sees the arguments*.
336
336
 
337
337
  This resolves one submitted value to one canonical value, and does not loosen the strict token match in [MCP-side list filtering](#mcp-side-list-filtering), which scores a query against many candidate names.
338
338
 
@@ -423,7 +423,7 @@ Two params, two behaviors — keep them named distinctly:
423
423
 
424
424
  Errors are part of the tool's interface — design them during the design phase, not as an afterthought. Three aspects: **the contract** (which failures are public), **classification** (what error code), and **messaging** (what the LLM reads).
425
425
 
426
- **Declare a typed contract for domain failures.** When a tool has known failure modes the agent should plan around (`no_match`, `queue_full`, `vendor_down`), enumerate them as `errors: [{ reason, code, when, recovery, retryable? }]` on the definition. `recovery` is required metadata — the agent's next move when this failure fires (≥ 5 words, lint-validated; spread `ctx.recoveryFor('reason')` into the throw-site `data` to send it on the wire as `data.recovery.hint`). The framework types `ctx.fail(reason, …)` against the declared reason union (typos become TS errors) and auto-populates `data.reason` on the thrown error for stable observability. The error reaches clients with parity across both surfaces — `structuredContent.error` (Claude Code) and `content[]` text (Claude Desktop). Baseline codes (`InternalError`, `ServiceUnavailable`, `Timeout`, `ValidationError`, `SerializationError`, `RequestCancelled`) bubble from anywhere and don't need to be enumerated. Mark an entry the service layer throws, rather than the handler, with `thrownBy: 'service'` so the conformance lint doesn't report it as a reason the handler never raises. See `api-errors` skill for the full pattern.
426
+ **Declare a typed contract for domain failures.** When a tool has known failure modes the agent should plan around (`no_match`, `queue_full`, `vendor_down`), enumerate them as `errors: [{ reason, code, when, recovery, retryable? }]` on the definition. `recovery` is required metadata — the agent's next move when this failure fires (≥ 5 words, lint-validated; the framework sends it on the wire as `data.recovery.hint` with any failure carrying that reason and no hint of its own). The framework types `ctx.fail(reason, …)` against the declared reason union (typos become TS errors) and auto-populates `data.reason` on the thrown error for stable observability. The error reaches clients with parity across both surfaces — `structuredContent.error` (Claude Code) and `content[]` text (Claude Desktop). Baseline codes (`InternalError`, `ServiceUnavailable`, `Timeout`, `ValidationError`, `SerializationError`, `RequestCancelled`) bubble from anywhere and don't need to be enumerated. Mark an entry the service layer throws, rather than the handler, with `thrownBy: 'service'` so the conformance lint doesn't report it as a reason the handler never raises. See `api-errors` skill for the full pattern.
427
427
 
428
428
  **Classify errors by origin.** Different error sources need different codes and different recovery guidance. Map the failure modes for each tool during design:
429
429
 
@@ -702,7 +702,7 @@ Items without an `If …:` prefix apply to every design. Conditional items only
702
702
  - [ ] **If the server has workflow tools:** call-flow documented (upstream sequence + mode arms) in design doc's Workflow Analysis
703
703
  - [ ] **If state-aware procedural guidance adds value:** instruction tool considered with `nextToolSuggestions` pre-filled from diagnostics
704
704
  - [ ] **If any tool is config-gated:** nothing routes to it while the gate is off — recovery strings, notices, and `guidance` name a callable target or state the capability is unavailable, and structured follow-ups naming it are emitted only under the config that registers it
705
- - [ ] **If workflow tools have destructive modes:** destructive arm gated on a `ctx.requestInput` confirmation read back from `ctx.inputs`, with `destructiveHint` annotation so clients that never fulfil the round still surface the risk
705
+ - [ ] **If workflow tools have destructive modes:** destructive arm gated on a `ctx.requestInput` confirmation that redeems a `ctx.state` consent record bound to the operation, caller, and target (shared storage when retries can reach another instance, `MCP_REQUEST_STATE_KEY` set, an arm that must not run twice idempotent per record), with `destructiveHint` annotation so clients that never fulfil the round still surface the risk
706
706
  - [ ] **If any tool calls `ctx.requestInput`:** `createApp()` declares `sessionMode` with `require: 'stateful'`
707
707
  - [ ] **If any output carries text other people wrote:** those fields listed, `format()` quotes or fences free text and flattens CR/LF in inline slots, and the server instructions say the content is data
708
708
  - [ ] **If a parameter determines blast radius:** safe default set (e.g., `mode: 'preview'`, `dryRun: true`, `confirmCount` required)
@@ -4,7 +4,7 @@ description: >
4
4
  Exercise tools, resources, and prompts against a live HTTP server via MCP JSON-RPC over curl. Starts the server, surfaces the catalog, runs real and adversarial inputs, measures every call (bytes, token estimate, wall-clock) and weighs the catalog, and produces a tight report with concrete findings and numbered follow-up options. Use after adding or modifying definitions, or when the user asks to test, try out, or verify their MCP surface.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "2.16"
7
+ version: "2.18"
8
8
  audience: external
9
9
  type: debug
10
10
  ---
@@ -19,7 +19,7 @@ Unit tests (`add-test` skill) verify handler logic with mocked context. Field te
19
19
 
20
20
  This skill drives an HTTP server because curl + JSON-RPC is the most reliable harness for shell-based agents. The same handlers run on both transports — only the framing differs — so HTTP exercises the full functional surface. Both HTTP session modes are covered: a durable `Mcp-Session-Id` session, and the sessionless initialization a `MCP_SESSION_MODE=stateless` server performs.
21
21
 
22
- **Stdio coverage is a boot check only — run this before Step 1.** Run `bun run rebuild && bun run start:stdio < /dev/null`, and confirm the startup logs look clean (banner, expected tool/resource counts, no errors/warnings, no missing-config gripes). Redirecting stdin is what ends the run: the server treats EOF as a shutdown signal, boots fully, then exits on its own, so the log also shows the graceful-shutdown path. Do not background it and reach for `pkill` — a pattern like `pkill -f dist/index.js` matches every other stdio MCP server on the machine, including the ones the calling agent's own session is connected to. Pino logs go to stderr in stdio mode (stdout is reserved for JSON-RPC), so they print straight to the terminal when you run interactively. No need to call tools over stdio — the HTTP pass already covered handler behavior.
22
+ **Stdio coverage is a boot check only — run this before Step 1.** Run `bun run rebuild && bun run start:stdio < /dev/null`, and confirm the startup logs look clean: the `Core services constructed — N tool(s) …` record lists every registered tool, resource, and prompt in its `tools` / `resources` / `prompts` fields — the message text shows only counts — and a definition missing from them was never passed to `createApp()`. No errors/warnings, no missing-config gripes. The emoji startup banner prints only to a terminal, so its absence from an agent's shell is not a finding. Redirecting stdin is what ends the run: the server treats EOF as a shutdown signal, boots fully, then exits on its own, so the log also shows the graceful-shutdown path. Do not background it and reach for `pkill` — a pattern like `pkill -f dist/index.js` matches every other stdio MCP server on the machine, including the ones the calling agent's own session is connected to. Pino logs go to stderr in stdio mode (stdout is reserved for JSON-RPC), so they print straight to the terminal when you run interactively. No need to call tools over stdio — the HTTP pass already covered handler behavior.
23
23
 
24
24
  ---
25
25
 
@@ -402,7 +402,7 @@ Treat any hit as a `ux` finding in the report. The authoring rule lives under *T
402
402
  |:------------------------------------------------|:-------------|
403
403
  | `include` / `fields` / `expand` / `view` / `projection` parameter | Field selection: non-default value renders requested fields |
404
404
  | Array return with `query` / `filter` inputs | Empty result: does response explain *why* (echo criteria, suggest broadening)? |
405
- | Identifier, code, or enum-ish input (an ID format, a classification code, a unit, a place name, a list the docs say may be comma-joined) | Value-variant tolerance: re-send the happy-path call with each obvious variant of that value — lowercase, the bare leaf of a hierarchical code, a common domain alias, a delimiter-joined list where an array is accepted, the spelled-out form of an abbreviated name. Pass is either outcome: the call succeeds, or it fails with an error naming the expected shape. A miss or a bare validation failure on a variant that maps one-to-one onto a valid value is a `ux` finding. Probe **values** — variants of the argument *key* name, and a JSON-stringified array as a value, are handled by the framework, not the server. |
405
+ | Identifier, code, or enum-ish input (an ID format, a classification code, a unit, a place name, a list the docs say may be comma-joined) | Value-variant tolerance: re-send the happy-path call with each obvious variant of that value — lowercase, the bare leaf of a hierarchical code, a common domain alias, a delimiter-joined list where an array is accepted, the spelled-out form of an abbreviated name. Pass is either outcome: the call succeeds, or it fails with an error naming the expected shape. A miss or a bare validation failure on a variant that maps one-to-one onto a valid value is a `ux` finding. Probe **values** — variants of the argument *key* name, and a JSON-stringified array or object or an integer sent for a string as a value, are handled by the framework, not the server. |
406
406
  | Batch / bulk input (arrays of IDs, multi-item ops) | Partial success: mix valid + invalid items |
407
407
  | `annotations.readOnlyHint: true` | Confirm no mutation happened |
408
408
  | `annotations.idempotentHint: true` | Call twice with same input — safe? |
@@ -436,7 +436,7 @@ When a call surprises you — slow, hangs, returns terse output, surfaces an unh
436
436
 
437
437
  - **`content[]` is an array of blocks — read all of them, never just `content[0]`.** A success result is assembled as `[...ctx.content media blocks, ...the format()/JSON domain render, ...the enrichment trailer]`. Everything the handler put on `ctx.enrich` — empty-result notices, totals, query echoes, truncation disclosure — renders in that trailer, a **separate trailing block**, not inside the `format()` block. Quoting `content[0].text` and reporting those fields as absent from `content[]` is a false parity gap; the suggested fix (render them in `format()` too) would double-render them. Dump `.result.content` in full before claiming drift.
438
438
  - Tool domain errors return `{result: {content: [...], isError: true}}` — they live in `result`, not `error`. Check `isError`, not the JSON-RPC error field.
439
- - **Tool error code/reason** rides on `result.structuredContent.error.{code, message, data?.reason}` — inspect that, not just the text. `data` is only spread when the handler threw an `McpError` (or `ZodError`); plain `throw new Error(...)` won't populate `data.reason`. Use `ctx.fail`-thrown errors when the contract reason matters. The text in `result.content[0].text` mirrors the message, adds `Recovery: <hint>` when `data.recovery.hint` says something the message does not already say, and closes with `(reason <reason> · not retryable)` for whichever of `data.reason` / `data.retryable` is present — the numeric code stays JSON-only.
439
+ - **Tool error code/reason** rides on `result.structuredContent.error.{code, message, data?.reason}` — inspect that, not just the text. `data` carries what the handler threw as an `McpError` (or a `ZodError`'s issues) plus the framework's `data.requestId`; plain `throw new Error(...)` won't populate `data.reason`. Use `ctx.fail`-thrown errors when the contract reason matters — a declared reason arrives with its contract `recovery` as `data.recovery.hint` even when the throw site passed none. The text in `result.content[0].text` mirrors the message, adds `Recovery: <hint>` when `data.recovery.hint` says something the message does not already say, and closes with `(reason <reason> · not retryable · request <id>)` for whichever of `data.reason` / `data.retryable` / `data.requestId` is present — the numeric code stays JSON-only. The request id matches the `requestId` on that call's server log records.
440
440
  - **Resource errors** are JSON-RPC-level — they appear in the top-level `error.{code, data.reason}` field, not inside `result`. Resource handlers re-throw rather than producing an `isError` envelope.
441
441
  - JSON-RPC `error` only appears for protocol issues (bad session, malformed envelope, unknown method).
442
442
  - `mcp_call` already strips SSE framing. Pipe to `jq` for readability.
@@ -500,7 +500,7 @@ End with:
500
500
 
501
501
  ## Checklist
502
502
 
503
- - [ ] Stdio boot check completed — `bun run rebuild && bun run start:stdio < /dev/null` shows clean startup (banner, expected counts, no errors) and a graceful shutdown on EOF
503
+ - [ ] Stdio boot check completed — `bun run rebuild && bun run start:stdio < /dev/null` shows clean startup (every expected definition listed in the `Core services constructed` record's `tools` / `resources` / `prompts` fields, no errors) and a graceful shutdown on EOF
504
504
  - [ ] HTTP server built and started; real port parsed from log
505
505
  - [ ] Session initialized (a stateless server returns an empty `sid` — still a pass); `notifications/initialized` sent; negotiated protocol version matches the requested one (a downgrade is a finding)
506
506
  - [ ] Catalog surfaced and presented; descriptions audited for leaks (implementation details, meta-coaching, consumer-aware phrasing)
@@ -4,7 +4,7 @@ description: >
4
4
  Land working-tree changes as logical commits — the work grouped by concern, topped by a release commit (version bump, changelog, regenerated artifacts). The work commits land first, then the version bump, verification, and the release commit on top. Stops at "committed locally on main" — or, when the project releases through a release PR, at "release branch pushed, PR open". No tag, no push to main, no publish: the release-and-publish skill merges, tags, and ships from here. Distilled from the git_wrapup_instructions protocol.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "1.26"
7
+ version: "1.27"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -144,8 +144,8 @@ When every concern is committed, `git status` is clean. That clean tree is what
144
144
  Every file that declares a version must be updated. Skip any file that doesn't exist in the project. For `@cyanheads/mcp-ts-core` projects:
145
145
 
146
146
  - `package.json` — `version`
147
- - `server.json` — top-level `version` AND every `packages[].version` entry
148
- - `manifest.json` (if present) — `version`. Verify `name` is the bare package name (e.g. `bls-mcp-server`, not `@cyanheads/bls-mcp-server`)
147
+ - `server.json` — top-level `version` AND every `packages[].version` entry. `lint:mcp` flags a mismatch at either level
148
+ - `manifest.json` (if present) — `version`. Packaging validation fails on a mismatch, and on a scoped `name` (use `bls-mcp-server`, not `@cyanheads/bls-mcp-server`)
149
149
  - `.claude-plugin/plugin.json` and `.codex-plugin/plugin.json` (if present) — `version`. Packaging validation fails on a mismatch; `.codex-plugin/mcp.json` is connection config and carries none
150
150
  - `README.md` — version badge. Packaging validation fails on a mismatch with `package.json`; a literal `-` in a prerelease is escaped as `--` (`Version-0.14.0--rc.1-`)
151
151
  - `CLAUDE.md` / `AGENTS.md` — if they pin a version string
@@ -194,15 +194,16 @@ Both scripts are idempotent — safe to run even if nothing changed.
194
194
 
195
195
  ### 7. Run the verification gate
196
196
 
197
- The stack being shipped must pass verification. Both must succeed:
197
+ The stack being shipped must pass verification. All must succeed:
198
198
 
199
199
  ```bash
200
200
  bun run devcheck
201
+ bun run rebuild
201
202
  bun run test:all # or `bun run test` if no test:all script exists
202
203
  bun run test:package # only if the script exists — NOT part of test:all
203
204
  ```
204
205
 
205
- **If either fails, halt.** Do not bypass verification to land the release commit.
206
+ **If any fails, halt.** Do not bypass verification to land the release commit.
206
207
 
207
208
  The work is already committed by this point, so the fix is a new commit on top of the stack, under step 3's conventions — never `git commit --amend`, never a rebase, reset, or any other rewrite of a commit the stack already carries. Land the fix, then re-run this step. The same holds when the gate passes but leaves the tree dirty: `devcheck` auto-fixes as it runs, and a formatter fix to a file committed in step 3 is a follow-up commit of its own, not something to fold into the release commit.
208
209
 
@@ -298,13 +299,14 @@ If the working tree isn't clean or the release commit isn't at HEAD, something w
298
299
 
299
300
  - [ ] Diff reviewed end-to-end before the first commit
300
301
  - [ ] Work concerns committed before the version bump — a version-bearing file a work concern also touches ships whole in that concern's commit, so the release commit brings it the version hunk alone
301
- - [ ] Version bumped in every declaring file (`package.json`, `server.json`, `manifest.json`, `.claude-plugin/plugin.json`, `.codex-plugin/plugin.json`, README badge, `CLAUDE.md`/`AGENTS.md` if they pin a version) — verify by command, not by eye: `v=$(jq -r .version package.json); grep -rl "$v" package.json server.json manifest.json .claude-plugin/plugin.json .codex-plugin/plugin.json README.md | wc -l` must equal the count of files that exist, and `grep -c "Version-$v-" README.md` must print `1`. `lint:packaging` checks the README badge against `package.json`, so a stale badge now fails `devcheck` instead of shipping unnoticed — the grep still catches a badge written in a shape the check skips
302
+ - [ ] Version bumped in every declaring file (`package.json`, `server.json`, `manifest.json`, `.claude-plugin/plugin.json`, `.codex-plugin/plugin.json`, README badge, `CLAUDE.md`/`AGENTS.md` if they pin a version) — `devcheck` flags a mismatch in `server.json` (both levels), `manifest.json`, both plugin manifests, and the README badge; step 4's straggler grep covers the docs and Dockerfile labels
302
303
  - [ ] GH issues addressed by this work commented with what landed (if working from GH issues)
303
304
  - [ ] Docs updated for any new or changed features
304
305
  - [ ] Changelog authored at `changelog/<major.minor>.x/<version>.md`
305
306
  - [ ] `CHANGELOG.md` rollup regenerated (`bun run changelog:build`)
306
307
  - [ ] `docs/tree.md` regenerated if structure changed (`bun run tree`)
307
308
  - [ ] `bun run devcheck` passes
309
+ - [ ] `bun run rebuild` succeeds
308
310
  - [ ] `bun run test:all` (or `test`) passes
309
311
  - [ ] `bun run test:package` passes, when the project defines it — it guards the public-export manifest and `test:all` does not run it
310
312
  - [ ] Release PR mode: stack committed on `release/<version>`, never on `main`
@@ -4,7 +4,7 @@ description: >
4
4
  Pick and run a multi-phase workflow that chains foundational task skills (`git-wrapup`, `release-and-publish`, `maintenance`, `field-test`, `setup`, etc.) end-to-end. Routes user intent to a workflow file under `workflows/` — greenfield builds, maintenance + release, field-test + fix, or known-work + release. Single source for the universal rules (no commits without authorization, no destructive git, no marketing language), the orchestrator posture (own the goal, ground sub-agents in primary sources, verify against the goal), and the sub-agent strategy (orient block, parallel fanout, isolation, normalization) that apply across every workflow. Sub-agents are an optional capability — workflows run linearly when fanout isn't available.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "1.11"
7
+ version: "1.12"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -68,8 +68,8 @@ The orchestrator owns the goals. Workflow phases are not "run skill X" — they
68
68
 
69
69
  Before running a phase (or spawning a sub-agent for it), write down four things:
70
70
 
71
- 1. **Goal** — the verifiable end state this phase must produce. Concrete and testable: "v0.5.2 tag exists at HEAD with structured-markdown annotation; `bun run devcheck` green; `npm view <pkg>@0.5.2` resolves." Not fuzzy: "ran the release-and-publish skill."
72
- 2. **Primary sources** — the specific files, GH issues, and reference docs the sub-agent must read directly. Inlining content into the prompt is a paraphrase that loses nuance; agents grounded in the source catch details the orchestrator's summary missed. For GH issues, instruct both `gh issue view N --comments` (the comment thread) and the timeline cross-reference query in the Orient block (what references the issue, including cross-repo) — the body alone misses both. The orchestrator reads these sources too (to construct the prompt), but that's prompt construction, not a substitute for the sub-agent reading them.
71
+ 1. **Goal** — the verifiable end state this phase must produce. Concrete and testable: "v0.5.2 tag exists at HEAD and passes `bun run release:github -- --check`; `bun run devcheck` green; `npm view <pkg>@0.5.2` resolves." Not fuzzy: "ran the release-and-publish skill."
72
+ 2. **Primary sources** — the specific files, GH issues, and reference docs the sub-agent must read directly. Inlining content into the prompt is a paraphrase that loses nuance; agents grounded in the source catch details the orchestrator's summary missed. For GH issues, instruct the three reads in the Orient block — `gh issue view N` (the body), `gh issue view N --comments` (the thread; without a TTY it prints no body), and the timeline cross-reference query (what references the issue, including cross-repo). The orchestrator reads these sources too (to construct the prompt), but that's prompt construction, not a substitute for the sub-agent reading them.
73
73
  3. **Path** — the Tier 1 skill(s) and steps that get to the goal. This is what gets handed to the sub-agent.
74
74
  4. **Verification** — the read-only checks that confirm the goal was hit. Defined upfront, not as an afterthought.
75
75
 
@@ -79,7 +79,7 @@ Why the framing matters:
79
79
  - **Sub-agent self-reports describe intent, not always reality.** A goal you wrote down beforehand is the falsification target — the sub-agent's report is a hypothesis to verify against it.
80
80
  - **Replanning is local.** When verification fails, the goal is unchanged; the orchestrator picks a different path (re-spawn with the failure context, re-slice the work, intervene directly). Phase rework doesn't cascade.
81
81
 
82
- **Inform without inlining.** An enhanced sub-agent prompt names the specific primary sources and the goal — it does NOT paraphrase them. "Review GH issue #123 (read it via `gh issue view 123 --comments`); the goal is X; verify with Y" is the right shape. Pasting the issue body into the prompt forces the sub-agent to work from a paraphrase. Let the sub-agent read the source and explore for additional context as needed.
82
+ **Inform without inlining.** An enhanced sub-agent prompt names the specific primary sources and the goal — it does NOT paraphrase them. "Review GH issue #123 (read it via `gh issue view 123` and `gh issue view 123 --comments`); the goal is X; verify with Y" is the right shape. Pasting the issue body into the prompt forces the sub-agent to work from a paraphrase. Let the sub-agent read the source and explore for additional context as needed.
83
83
 
84
84
  ## Sub-Agent Strategy (if available)
85
85
 
@@ -122,8 +122,9 @@ order. If any file does not exist, note it and continue.
122
122
  5. Read the skill file(s) for this task: `[Tier 1 skill paths]`.
123
123
  6. Read the primary sources for this task directly — design docs (`docs/design.md`),
124
124
  GH issues, handoff documents, reference/gold-standard files. For a GH issue, read
125
- both the comment thread and its cross-references — the body alone misses both:
126
- - `gh issue view <N> --comments` — description + comment thread
125
+ the body, the comment thread, and its cross-references — three separate reads:
126
+ - `gh issue view <N>` — the body
127
+ - `gh issue view <N> --comments` — the comment thread (without a TTY it prints no body, so it never replaces the read above)
127
128
  - `gh api 'repos/{owner}/{repo}/issues/<N>/timeline' --paginate --jq '.[] | select(.event=="cross-referenced") | .source.issue | "\(.repository.full_name)#\(.number) — \(.title)"'` — issues/PRs that reference this one, including from other repos
128
129
  List each source explicitly: `[primary source paths and gh commands]`. Skip this
129
130
  step only if no primary source applies (rare).
@@ -4,7 +4,7 @@ description: >
4
4
  Workflow: field-test one or more existing MCP server projects against the live upstream API, file GH issues for valid findings, deploy fix sub-agents per server, optionally loop until clean, then wrap up and release. Chains the `field-test`, `report-issue-local`, `tool-defs-analysis`, `code-simplifier`, `git-wrapup`, and `release-and-publish` skills. Read `../SKILL.md` first for the universal rules and sub-agent strategy.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "1.1"
7
+ version: "1.2"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -56,8 +56,8 @@ Each phase's Objective column is the goal state per target — the verifiable en
56
56
  | 3 | Fix | Per target: priority issues fixed in source, tests updated, `devcheck` + `test` green, each issue commented with fix details, working tree dirty for review | parallel fanout (one sub-agent per target — hard constraint) | gate-free |
57
57
  | 4 | Verify | Per target: full diff cold-reviewed; simplified if warranted; each fix re-exercised against the running server with actual tool output in the summary | parallel fanout | **barrier** — orchestrator loop decision (human/evidence-based: proceed, loop, or surface to user) |
58
58
  | 5 | Loop decision | Orchestrator decision recorded — proceed to release, loop another field-test cycle, or pause/surface to user. Evidence-based | orchestrator (serial) | **barrier** — release authorization required before advancing |
59
- | 6 | Wrap-up + release | (Optional) Per target: fixes split into per-file commits with a release commit on top; annotated tag; published per repo visibility; tag annotation is structured markdown with issue backlinks | parallel fanout (Bash git only) | gate-free |
60
- | 7 | Issue cleanup | Every GH issue that shipped a fix closed with "Fixed in v\<version\>" comment; skipped issues remain open | orchestrator (serial) | — |
59
+ | 6 | Wrap-up + release | (Optional) Per target: fixes grouped into one commit per concern (a file never splits across commits) with a release commit on top; annotated tag passing `bun run release:github -- --check` (flat bullets with issue backlinks, changelog link last); published per repo visibility | parallel fanout (Bash git only) | gate-free |
60
+ | 7 | Issue cleanup | Every GH issue that shipped a fix closed (reason: completed) carrying exactly one what-landed comment that cites the version; skipped issues remain open | orchestrator (serial) | — |
61
61
 
62
62
  Phase 6 is optional — stop earlier if release isn't authorized. Phase 7 only runs if Phase 6 ran.
63
63
 
@@ -92,7 +92,7 @@ Orchestrator verifies filed issues exist via `gh issue list -R <owner>/<repo>` p
92
92
  **One sub-agent per target — hard constraint.** No file-locking system exists for concurrent edits; multiple agents touching the same server's `src/` will conflict.
93
93
 
94
94
  Each sub-agent:
95
- 1. Reads all open issues for its target via `gh issue list` + `gh issue view N --comments` (full thread — body alone misses clarifications)
95
+ 1. Reads all open issues for its target via `gh issue list`, then `gh issue view N` and `gh issue view N --comments` per issue (body, then thread — the body alone misses clarifications)
96
96
  2. **Validates each issue against source code** — a "fixed" issue is a misdiagnosed one if validation fails
97
97
  3. Implements fixes in priority order: security → bugs → UX
98
98
  4. Rebuilds after each fix or group of related fixes
@@ -143,19 +143,7 @@ The changelog carries the depth; the tag annotation covers every change at headl
143
143
 
144
144
  **Version bump.** Default **patch** for field-test fix releases. **Minor** when enhancements are bundled in.
145
145
 
146
- **Tag annotation format.** Tag subject omits the version number. Structured markdown:
147
-
148
- ```
149
- Field-test bug fixes across N tools
150
-
151
- Fixed:
152
- - <tool_name>: <one-line fix description> (#<issue>)
153
- - <tool_name>: <one-line fix description> (#<issue>)
154
-
155
- <test count>; `bun run devcheck` clean.
156
- ```
157
-
158
- Add a `Security:` section when the changelog frontmatter sets `security: true`.
146
+ **Tag annotation.** Written at release time in the `release-and-publish` step 4 format — a short theme subject, flat bullets with `(#N)` backlinks, the changelog link last; `bun run release:github -- --check` enforces the shape before the push.
159
147
 
160
148
  **Wrap-up scope.** Determined by repo visibility:
161
149
 
@@ -167,9 +155,11 @@ Add a `Security:` section when the changelog frontmatter sets `security: true`.
167
155
  ### Phase 7: Issue cleanup
168
156
  Close issues that shipped fixes — only those. Skipped issues stay open.
169
157
 
158
+ Each issue carries exactly ONE what-landed comment — the fix summary Phase 3 posted, with the version added (`Shipped in v<version>: …`) if it lacks one. Then close without an additional comment:
159
+
170
160
  ```bash
171
161
  for n in <fixed-issue-numbers-from-phase-3>; do
172
- gh issue close "$n" -R "<owner>/<repo>" --reason completed --comment "Fixed in v<version>."
162
+ gh issue close "$n" -R "<owner>/<repo>" --reason completed
173
163
  done
174
164
  ```
175
165
 
@@ -205,4 +195,4 @@ Collect specific issue numbers from Phase 3 sub-agent summaries — do not close
205
195
  - [ ] Phase 6 (if releasing): version bumped, fix commits + release commit, annotated tag, scope matches private/public status
206
196
  - [ ] Phase 7 (if releasing): fixed issues closed; skipped issues remain open
207
197
  - [ ] Post-workflow verification: `git ls-remote --tags origin`, `npm view <pkg>@<version>` if public, GH release artifacts attached
208
- - [ ] Tag/release quality review: tag subject omits version number, structured markdown, no marketing adjectives, issue backlinks present
198
+ - [ ] Tag/release quality review: `bun run release:github -- --check` passed before the push; no marketing adjectives, issue backlinks present
@@ -4,7 +4,7 @@ description: >
4
4
  Workflow for landing known work (handoff document findings, tracked GH issues, observed gaps) and shipping it: fix → optional simplify and field-test verification → wrap-up → release across one or more MCP server projects. Generalizes "I have known issues to fix and ship" regardless of how the issues were surfaced. Chains the `field-test`, `report-issue-local`, `code-simplifier`, `git-wrapup`, and `release-and-publish` skills. Read `../SKILL.md` first for the universal rules and sub-agent strategy.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "1.1"
7
+ version: "1.2"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -20,7 +20,7 @@ The input varies but the workflow is the same. Read the inputs into a common sha
20
20
  | Source | Shape it as |
21
21
  |:---|:---|
22
22
  | Handoff document (numbered findings, repro steps, acceptance criteria) | Validate each finding live in Phase 1a; file each valid one as a GH issue via `report-issue-local`; skip invalidated findings |
23
- | GH issues already filed | Use as-is. Read each with `gh issue view N --comments` to capture the full thread (the body alone misses clarifications and decision updates) |
23
+ | GH issues already filed | Use as-is. Read each with `gh issue view N` (the body) and `gh issue view N --comments` (the thread — the body alone misses clarifications and decision updates) |
24
24
  | Observed gap or casual report ("I noticed this", "fix the description on tool X") | If material enough to ship in a release, file a GH issue first to capture rationale and create an audit trail. Trivial typo-fix-and-ship can skip the issue step. |
25
25
 
26
26
  The validation/filing step is the difference between "input is a hypothesis" (handoff) and "input is verified" (tracked GH issues). The rest of the workflow is identical.
@@ -47,7 +47,7 @@ For unsourced QA — where the bugs are unknown until you test — use `field-te
47
47
 
48
48
  Per target:
49
49
 
50
- 1. **Identify issues** — collect GH issue numbers to fix, the handoff document, or the explicit gap description. Read each issue with `gh issue view N --comments` to capture the full thread.
50
+ 1. **Identify issues** — collect GH issue numbers to fix, the handoff document, or the explicit gap description. Read each issue with `gh issue view N` and `gh issue view N --comments` — body, then thread.
51
51
  2. **Clean working tree** — `git status --short` must be empty
52
52
  3. **Current version** — `git describe --tags --abbrev=0`, `grep '"version"' package.json`
53
53
  4. **Repo visibility** — `gh repo view --json visibility -q '.visibility'`. Determines wrap-up scope.
@@ -62,7 +62,7 @@ Each phase's Objective column is the goal state per target — the verifiable en
62
62
  | 1a | Validate (conditional) | Each handoff finding field-tested live; valid ones filed as GH issues; invalidated ones reported back with reason. If zero validate, workflow stops | one sub-agent per target | **barrier** — cross-target synthesis: orchestrator confirms validated findings before fix proceeds (or stops workflow if zero validate) |
63
63
  | 1b | Fix | Per target: targeted issues fixed in source, tests updated/added, `devcheck` + `rebuild` + `test` green, each fixed issue commented with fix details, working tree dirty for review | parallel fanout (one sub-agent per target — hard constraint) | **barrier** — orchestrator reviews diffs before verify (explicit gate in checklist) |
64
64
  | 2 | Verify | Per target: full diff cold-reviewed; simplified if warranted; each fix re-exercised against the running server with actual tool output in the summary | parallel fanout | **barrier** — orchestrator reviews simplified diff and verified outputs; release authorization required |
65
- | 3 | Wrap-up + release | Per target: fixes split into per-file commits with a release commit on top; annotated tag; published per repo visibility; tag annotation is structured markdown with issue backlinks | parallel fanout (Bash git only) | gate-free |
65
+ | 3 | Wrap-up + release | Per target: fixes grouped into one commit per concern (a file never splits across commits) with a release commit on top; annotated tag passing `bun run release:github -- --check` (flat bullets with issue backlinks, changelog link last); published per repo visibility | parallel fanout (Bash git only) | gate-free |
66
66
  | 4 | Issue cleanup | Every shipped issue closed (reason: completed) carrying exactly one what-landed comment that cites the version | orchestrator (serial) | — |
67
67
 
68
68
  Phase 1a is conditional — only runs when the input is a handoff document or otherwise unvalidated. When the input is already tracked GH issues, skip directly to Phase 1b. The release portion of Phase 3 is conditional on user authorization to ship.
@@ -89,7 +89,7 @@ If zero findings validate, report to the user and stop the workflow.
89
89
  **One sub-agent per target — hard constraint** (no file-locking; concurrent edits to the same `src/` conflict).
90
90
 
91
91
  Each sub-agent:
92
- 1. Reads all open issues for its target via `gh issue view N --comments` (full thread — body alone misses clarifications)
92
+ 1. Reads all open issues for its target via `gh issue view N` and `gh issue view N --comments` (body, then thread — the body alone misses clarifications)
93
93
  2. **Validates each issue against source code** — the issue's analysis or proposed approach may be wrong; sub-agent applies judgment about the right fix and notes any deviation in its GH comment
94
94
  3. Prioritizes: security → crashes → bugs → enhancements → docs/chore
95
95
  4. Implements fixes using the best modern approach (the GH issue is input, not a spec)
@@ -162,7 +162,7 @@ If no what-landed comment exists yet, the version belongs in that one comment ("
162
162
  | 5 | Code-simplify removes intentional complexity | Orchestrator gate after Phase 2 reviews the full diff |
163
163
  | 6 | Wrap-up sub-agent collapses multi-fix diff into one commit | Phase 3 prompt enumerates the commit structure |
164
164
  | 7 | Wrap-up sub-agent makes unplanned intermediate commits outside the planned structure | Prompt defines exact commit shape; agents must not invent extras |
165
- | 8 | Reading `gh issue view N` alone misses thread context where decisions were updated | Always include `--comments` |
165
+ | 8 | Reading `gh issue view N` alone misses thread context where decisions were updated; `--comments` alone prints no body without a TTY | Always run both |
166
166
  | 9 | MCP Registry returns 502 transiently during publish | Retry up to 2x with backoff |
167
167
  | 10 | Phase 1a sub-agent validates an issue that's actually a misunderstanding | Sub-agent must field-test, not just read the claim — live verification catches false positives |
168
168
 
@@ -178,4 +178,4 @@ If no what-landed comment exists yet, the version belongs in that one comment ("
178
178
  - [ ] Phase 3: published per scope (push, npm if public, MCP Registry if applicable, GH release, Docker if applicable)
179
179
  - [ ] Phase 4: shipped issues closed, one what-landed comment each; skipped issues remain open
180
180
  - [ ] Post-workflow verification: `git ls-remote --tags origin`, `npm view <pkg>@<version>` if public, GH release artifacts attached
181
- - [ ] Tag/release quality review: tag subject omits version number, structured markdown, no marketing adjectives, issue backlinks present
181
+ - [ ] Tag/release quality review: `bun run release:github -- --check` passed before the push; no marketing adjectives, issue backlinks present