@cyanheads/mcp-ts-core 0.13.8 → 0.13.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +51 -24
- package/CLAUDE.md +51 -24
- package/README.md +10 -10
- package/changelog/0.13.x/0.13.10.md +118 -0
- package/changelog/0.13.x/0.13.9.md +113 -0
- package/dist/config/index.d.ts +3 -0
- package/dist/config/index.d.ts.map +1 -1
- package/dist/config/index.js +31 -9
- package/dist/config/index.js.map +1 -1
- package/dist/core/app.d.ts +6 -3
- package/dist/core/app.d.ts.map +1 -1
- package/dist/core/app.js +21 -5
- package/dist/core/app.js.map +1 -1
- package/dist/core/context.d.ts +114 -21
- package/dist/core/context.d.ts.map +1 -1
- package/dist/core/context.js +40 -0
- package/dist/core/context.js.map +1 -1
- package/dist/core/index.d.ts +1 -1
- package/dist/core/index.d.ts.map +1 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/serverManifest.d.ts +6 -0
- package/dist/core/serverManifest.d.ts.map +1 -1
- package/dist/core/serverManifest.js +6 -0
- package/dist/core/serverManifest.js.map +1 -1
- package/dist/core/worker.d.ts +6 -0
- package/dist/core/worker.d.ts.map +1 -1
- package/dist/core/worker.js +1 -0
- package/dist/core/worker.js.map +1 -1
- package/dist/linter/rules/error-contract-rules.d.ts +3 -44
- package/dist/linter/rules/error-contract-rules.d.ts.map +1 -1
- package/dist/linter/rules/error-contract-rules.js +8 -144
- package/dist/linter/rules/error-contract-rules.js.map +1 -1
- package/dist/linter/rules/index.d.ts +1 -1
- package/dist/linter/rules/index.d.ts.map +1 -1
- package/dist/linter/rules/index.js +1 -1
- package/dist/linter/rules/index.js.map +1 -1
- package/dist/linter/rules/resource-rules.d.ts.map +1 -1
- package/dist/linter/rules/resource-rules.js +1 -2
- package/dist/linter/rules/resource-rules.js.map +1 -1
- package/dist/linter/rules/tool-rules.d.ts +2 -1
- package/dist/linter/rules/tool-rules.d.ts.map +1 -1
- package/dist/linter/rules/tool-rules.js +37 -3
- package/dist/linter/rules/tool-rules.js.map +1 -1
- package/dist/mcp-server/handlerContext.d.ts +26 -13
- package/dist/mcp-server/handlerContext.d.ts.map +1 -1
- package/dist/mcp-server/handlerContext.js +32 -17
- package/dist/mcp-server/handlerContext.js.map +1 -1
- package/dist/mcp-server/inputRequired.d.ts +133 -12
- package/dist/mcp-server/inputRequired.d.ts.map +1 -1
- package/dist/mcp-server/inputRequired.js +192 -20
- package/dist/mcp-server/inputRequired.js.map +1 -1
- package/dist/mcp-server/prompts/prompt-registration.d.ts +10 -2
- package/dist/mcp-server/prompts/prompt-registration.d.ts.map +1 -1
- package/dist/mcp-server/prompts/prompt-registration.js +49 -11
- package/dist/mcp-server/prompts/prompt-registration.js.map +1 -1
- package/dist/mcp-server/resources/resource-registration.d.ts +4 -2
- package/dist/mcp-server/resources/resource-registration.d.ts.map +1 -1
- package/dist/mcp-server/resources/resource-registration.js +6 -4
- package/dist/mcp-server/resources/resource-registration.js.map +1 -1
- package/dist/mcp-server/resources/utils/resourceHandlerFactory.d.ts +4 -3
- package/dist/mcp-server/resources/utils/resourceHandlerFactory.d.ts.map +1 -1
- package/dist/mcp-server/resources/utils/resourceHandlerFactory.js +42 -14
- package/dist/mcp-server/resources/utils/resourceHandlerFactory.js.map +1 -1
- package/dist/mcp-server/server.d.ts +9 -0
- package/dist/mcp-server/server.d.ts.map +1 -1
- package/dist/mcp-server/server.js +14 -13
- package/dist/mcp-server/server.js.map +1 -1
- package/dist/mcp-server/tools/tool-registration.d.ts +7 -3
- package/dist/mcp-server/tools/tool-registration.d.ts.map +1 -1
- package/dist/mcp-server/tools/tool-registration.js +9 -5
- package/dist/mcp-server/tools/tool-registration.js.map +1 -1
- package/dist/mcp-server/tools/utils/inputPrevalidation.d.ts +163 -40
- package/dist/mcp-server/tools/utils/inputPrevalidation.d.ts.map +1 -1
- package/dist/mcp-server/tools/utils/inputPrevalidation.js +330 -114
- package/dist/mcp-server/tools/utils/inputPrevalidation.js.map +1 -1
- package/dist/mcp-server/tools/utils/toolHandlerFactory.d.ts +40 -19
- package/dist/mcp-server/tools/utils/toolHandlerFactory.d.ts.map +1 -1
- package/dist/mcp-server/tools/utils/toolHandlerFactory.js +387 -130
- package/dist/mcp-server/tools/utils/toolHandlerFactory.js.map +1 -1
- package/dist/mcp-server/transports/http/httpErrorHandler.d.ts.map +1 -1
- package/dist/mcp-server/transports/http/httpErrorHandler.js +7 -1
- package/dist/mcp-server/transports/http/httpErrorHandler.js.map +1 -1
- package/dist/mcp-server/transports/http/httpTransport.d.ts.map +1 -1
- package/dist/mcp-server/transports/http/httpTransport.js +65 -9
- package/dist/mcp-server/transports/http/httpTransport.js.map +1 -1
- package/dist/mcp-server/transports/stdio/stdioTransport.d.ts +9 -5
- package/dist/mcp-server/transports/stdio/stdioTransport.d.ts.map +1 -1
- package/dist/mcp-server/transports/stdio/stdioTransport.js +9 -5
- package/dist/mcp-server/transports/stdio/stdioTransport.js.map +1 -1
- package/dist/services/canvas/core/CanvasRegistry.d.ts +6 -2
- package/dist/services/canvas/core/CanvasRegistry.d.ts.map +1 -1
- package/dist/services/canvas/core/CanvasRegistry.js +7 -3
- package/dist/services/canvas/core/CanvasRegistry.js.map +1 -1
- package/dist/services/canvas/providers/duckdb/DuckdbProvider.d.ts +82 -18
- package/dist/services/canvas/providers/duckdb/DuckdbProvider.d.ts.map +1 -1
- package/dist/services/canvas/providers/duckdb/DuckdbProvider.js +620 -328
- package/dist/services/canvas/providers/duckdb/DuckdbProvider.js.map +1 -1
- package/dist/services/canvas/providers/duckdb/exportWriter.d.ts +11 -7
- package/dist/services/canvas/providers/duckdb/exportWriter.d.ts.map +1 -1
- package/dist/services/canvas/providers/duckdb/exportWriter.js +19 -16
- package/dist/services/canvas/providers/duckdb/exportWriter.js.map +1 -1
- package/dist/services/mirror/core/defineMirror.d.ts +1 -0
- package/dist/services/mirror/core/defineMirror.d.ts.map +1 -1
- package/dist/services/mirror/core/defineMirror.js +1 -0
- package/dist/services/mirror/core/defineMirror.js.map +1 -1
- package/dist/testing/index.d.ts +17 -2
- package/dist/testing/index.d.ts.map +1 -1
- package/dist/testing/index.js +21 -7
- package/dist/testing/index.js.map +1 -1
- package/dist/types-global/errors.d.ts +18 -15
- package/dist/types-global/errors.d.ts.map +1 -1
- package/dist/utils/index.d.ts +1 -1
- package/dist/utils/index.d.ts.map +1 -1
- package/dist/utils/index.js.map +1 -1
- package/dist/utils/internal/error-handler/errorHandler.d.ts +5 -4
- package/dist/utils/internal/error-handler/errorHandler.d.ts.map +1 -1
- package/dist/utils/internal/error-handler/errorHandler.js +9 -7
- package/dist/utils/internal/error-handler/errorHandler.js.map +1 -1
- package/dist/utils/internal/error-handler/types.d.ts +3 -1
- package/dist/utils/internal/error-handler/types.d.ts.map +1 -1
- package/dist/utils/internal/performance.d.ts +4 -2
- package/dist/utils/internal/performance.d.ts.map +1 -1
- package/dist/utils/internal/performance.js +8 -6
- package/dist/utils/internal/performance.js.map +1 -1
- package/dist/utils/internal/telemetryMessages.d.ts +0 -1
- package/dist/utils/internal/telemetryMessages.d.ts.map +1 -1
- package/dist/utils/internal/telemetryMessages.js +0 -1
- package/dist/utils/internal/telemetryMessages.js.map +1 -1
- package/dist/utils/network/pacer.d.ts +38 -5
- package/dist/utils/network/pacer.d.ts.map +1 -1
- package/dist/utils/network/pacer.js +87 -25
- package/dist/utils/network/pacer.js.map +1 -1
- package/dist/utils/telemetry/attributes.d.ts +10 -5
- package/dist/utils/telemetry/attributes.d.ts.map +1 -1
- package/dist/utils/telemetry/attributes.js +10 -5
- package/dist/utils/telemetry/attributes.js.map +1 -1
- package/framework-skills/add-app-tool/SKILL.md +3 -3
- package/framework-skills/add-export/SKILL.md +5 -16
- package/framework-skills/add-prompt/SKILL.md +7 -3
- package/framework-skills/add-resource/SKILL.md +7 -5
- package/framework-skills/add-service/SKILL.md +3 -12
- package/framework-skills/add-test/SKILL.md +6 -3
- package/framework-skills/add-tool/SKILL.md +40 -42
- package/framework-skills/api-auth/SKILL.md +2 -2
- package/framework-skills/api-canvas/SKILL.md +17 -8
- package/framework-skills/api-config/SKILL.md +5 -4
- package/framework-skills/api-context/SKILL.md +168 -42
- package/framework-skills/api-errors/SKILL.md +48 -51
- package/framework-skills/api-linter/SKILL.md +30 -35
- package/framework-skills/api-mirror/SKILL.md +2 -1
- package/framework-skills/api-telemetry/SKILL.md +14 -10
- package/framework-skills/api-testing/SKILL.md +43 -11
- package/framework-skills/api-utils/SKILL.md +2 -2
- package/framework-skills/api-workers/SKILL.md +3 -1
- package/framework-skills/design-mcp-server/SKILL.md +6 -6
- package/framework-skills/field-test/SKILL.md +5 -5
- package/framework-skills/git-wrapup/SKILL.md +8 -6
- package/framework-skills/orchestrations/SKILL.md +7 -6
- package/framework-skills/orchestrations/workflows/field-test-fix.md +9 -19
- package/framework-skills/orchestrations/workflows/fix-wrapup-release.md +7 -7
- package/framework-skills/orchestrations/workflows/greenfield-build.md +8 -5
- package/framework-skills/orchestrations/workflows/maintenance-release.md +8 -8
- package/framework-skills/polish-docs-meta/SKILL.md +4 -4
- package/framework-skills/release-and-publish/SKILL.md +8 -6
- package/framework-skills/release-pr-review/SKILL.md +38 -24
- package/framework-skills/report-issue-framework/SKILL.md +7 -5
- package/framework-skills/report-issue-local/SKILL.md +8 -6
- package/framework-skills/security-pass/SKILL.md +14 -13
- package/package.json +6 -5
- package/scripts/devcheck.ts +7 -6
- package/scripts/install-otel.ts +84 -0
- package/scripts/lint-mcp.ts +87 -27
- package/scripts/lint-packaging.ts +226 -4
- package/scripts/release-github.ts +117 -5
- package/templates/.env.example +2 -0
- package/templates/AGENTS.md +5 -4
- package/templates/CLAUDE.md +5 -4
- package/templates/Dockerfile +67 -50
- package/templates/_.mcpbignore +2 -0
- package/templates/src/mcp-server/tools/definitions/echo.tool.ts +3 -8
|
@@ -4,7 +4,7 @@ description: >
|
|
|
4
4
|
Testing patterns for MCP tool/resource handlers using `createMockContext` and Vitest. Covers mock context options, handler testing, McpError assertions, format testing, Vitest config setup, and test isolation conventions.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "1.
|
|
7
|
+
version: "1.13"
|
|
8
8
|
audience: external
|
|
9
9
|
type: reference
|
|
10
10
|
---
|
|
@@ -133,6 +133,8 @@ toolContractSuite(searchTool, {
|
|
|
133
133
|
|
|
134
134
|
Use `runToolContract(definition, input, { context })` from `/testing` when a custom test runner or an imperative assertion is a better fit. It intentionally skips transport auth and telemetry; those belong in transport/integration tests.
|
|
135
135
|
|
|
136
|
+
A declared reason thrown without a hint — a bare `ctx.fail('reason')` or a service throw carrying `{ reason }` — comes back with the entry's `recovery` as `data.recovery.hint` and a `Recovery:` line in `content[]`, as in production. The one production field it leaves out is `data.requestId` (and the `request <id>` term closing `content[]`), since there is no real request; a test asserting the factory's envelope instead expects both. Calling `definition.handler(...)` directly returns the `McpError` exactly as the throw site built it — no fill, no request id.
|
|
137
|
+
|
|
136
138
|
Arguments that fail the `input` schema are rejected the way the production handler factory rejects them: `InvalidParams` (`-32602`), with a message naming the tool and every failing field. That is the code a client sees on the wire, so assert it — not `ValidationError` (`-32007`), which stays the classification for a `ZodError` a handler throws itself. A result that breaks the tool's own `output` or `enrichment` schema is the definition's bug, so it returns `InternalError` (`-32603`) with a message naming that contract, exactly as in production.
|
|
137
139
|
|
|
138
140
|
Cancellation settles as it does in production. Pass `context: { signal }` and abort it: once the signal has fired, whatever the handler — or the output validation, `format()`, and enrichment after it — throws comes back as `RequestCancelled` (`-32011`), whether that is the signal's `AbortError`, its reason string, a `withRetry` backoff that stopped, or an `McpError` of the handler's own. A throw while the signal is still live keeps its own classification, and argument parsing stays outside the settle, so schema-invalid arguments on an aborted signal still return `InvalidParams`. A `toolContractSuite` error case with an aborted `context.signal` asserts `code: JsonRpcErrorCode.RequestCancelled` the same way.
|
|
@@ -149,6 +151,7 @@ createMockContext({ tenantId: 'test-tenant' }) // explicit tenant
|
|
|
149
151
|
createMockContext({ errors: myTool.errors }) // attaches typed ctx.fail keyed by the contract reasons
|
|
150
152
|
createMockContext({ inputResponses: { confirm: { action: 'accept', content: { ok: true } } } }) // second round of a multi-round-trip handler
|
|
151
153
|
createMockContext({ requestState: 'opaque-state' }) // seeds ctx.inputs.state()
|
|
154
|
+
createMockContext({ clientCapabilities: { roots: {} } }) // seeds ctx.clientCapabilities and filters inputResponses to declared kinds
|
|
152
155
|
createMockContext({ requestId: 'my-id' }) // override request ID (default: 'test-request-id')
|
|
153
156
|
createMockContext({ notifyResourceListChanged: () => {} }) // with resource-list change notifier
|
|
154
157
|
createMockContext({ notifyResourceUpdated: (_uri) => {} }) // with resource update notifier
|
|
@@ -162,6 +165,7 @@ createMockContext({ uri: new URL('myscheme://item/123') }) // for resource han
|
|
|
162
165
|
```ts
|
|
163
166
|
interface MockContextOptions<TErrors extends readonly ErrorContract[] | undefined> {
|
|
164
167
|
auth?: AuthContext;
|
|
168
|
+
clientCapabilities?: ClientCapabilities;
|
|
165
169
|
errors?: TErrors | undefined;
|
|
166
170
|
inputResponses?: InputResponses | Record<string, unknown>;
|
|
167
171
|
notifyPromptListChanged?: () => void;
|
|
@@ -181,6 +185,7 @@ interface MockContextOptions<TErrors extends readonly ErrorContract[] | undefine
|
|
|
181
185
|
|:-------|:-------|
|
|
182
186
|
| _(none)_ | Working `ctx.state` on tenant `'default'`; `ctx.inputs` is empty (first round) |
|
|
183
187
|
| `auth` | Sets `ctx.auth` for scope-checking tests |
|
|
188
|
+
| `clientCapabilities` | Sets `ctx.clientCapabilities` (`undefined` when omitted) and applies the production filter to `inputResponses`: only the answers these capabilities cover reach `ctx.inputs` (elicit → `elicitation`, and `elicitation.form` when it carries `content`; sampling → `sampling`, and `sampling.tools` when it holds a `tool_use` / `tool_result` block; roots → `roots`). Omitted, every seeded response reaches `ctx.inputs`, so existing `{ inputResponses }` tests are unaffected. The mock's `ctx.requestInput` stays ungated either way |
|
|
184
189
|
| `errors` | Attaches a typed `ctx.fail` against the contract — same wiring the production handler factory uses. Pass `myTool.errors` directly; the return type narrows to `HandlerContext<ReasonOf<…>>`, so the context is assignable to that definition's handler parameter. |
|
|
185
190
|
| `inputResponses` | Seeds `ctx.inputs` with the responses a retried request would carry, keyed by the identifiers the handler's `ctx.requestInput(...)` assigned (see below) |
|
|
186
191
|
| `notifyPromptListChanged` | Assigns `ctx.notifyPromptListChanged` for prompt-list change notification tests |
|
|
@@ -219,14 +224,14 @@ Reach for `createInMemoryStorage()` when a service takes a `StorageService` dire
|
|
|
219
224
|
|
|
220
225
|
### Mock inputs
|
|
221
226
|
|
|
222
|
-
`ctx.requestInput` is the real implementation: it throws an `InputRequiredSignal` the production handler factories convert into an `input_required` result. In a unit test the handler is called directly, so that signal surfaces as a thrown value — which is exactly how you assert the first round.
|
|
227
|
+
`ctx.requestInput` is the real implementation: it throws an `InputRequiredSignal` the production handler factories convert into an `input_required` result. In a unit test the handler is called directly, so that signal surfaces as a thrown value — which is exactly how you assert the first round. The examples below drive `export_report` from `api-context` § *The shape of a multi-round-trip handler*, which asks for a format the caller left out:
|
|
223
228
|
|
|
224
229
|
```ts
|
|
225
230
|
import { isInputRequiredSignal } from '@cyanheads/mcp-ts-core';
|
|
226
231
|
|
|
227
|
-
it('asks for
|
|
232
|
+
it('asks for the format on the first round', async () => {
|
|
228
233
|
const ctx = createMockContext();
|
|
229
|
-
await expect(
|
|
234
|
+
await expect(exportReport.handler(exportReport.input.parse({ reportId: 'r1' }), ctx))
|
|
230
235
|
.rejects.toSatisfy(isInputRequiredSignal);
|
|
231
236
|
});
|
|
232
237
|
```
|
|
@@ -236,30 +241,57 @@ To assert on *what* was requested, use `expectInputRequired` from `/testing`. It
|
|
|
236
241
|
```ts
|
|
237
242
|
import { createMockContext, expectInputRequired } from '@cyanheads/mcp-ts-core/testing';
|
|
238
243
|
|
|
239
|
-
const asked = await expectInputRequired(() =>
|
|
240
|
-
expect(asked.inputRequests?.
|
|
244
|
+
const asked = await expectInputRequired(() => exportReport.handler(input, createMockContext()));
|
|
245
|
+
expect(asked.inputRequests?.format?.method).toBe('elicitation/create');
|
|
241
246
|
```
|
|
242
247
|
|
|
243
248
|
Pass `asked.requestState` back as `createMockContext({ requestState })` when the handler reads state from the prior round. `inputResponses` drives the second round. `ctx.inputs.accepted(key, schema)` and `.view(key)` read it with the same helpers production uses, so a wrong response shape fails in the test:
|
|
244
249
|
|
|
245
250
|
```ts
|
|
246
|
-
it('
|
|
251
|
+
it('exports in the format the user picked', async () => {
|
|
247
252
|
const ctx = createMockContext({
|
|
248
|
-
inputResponses: {
|
|
253
|
+
inputResponses: { format: { action: 'accept', content: { format: 'csv' } } },
|
|
249
254
|
});
|
|
250
|
-
await expect(
|
|
255
|
+
await expect(exportReport.handler(input, ctx)).resolves.toMatchObject({ url: expect.any(String) });
|
|
251
256
|
});
|
|
252
257
|
|
|
253
258
|
it('stops when the user declines', async () => {
|
|
254
259
|
const ctx = createMockContext({
|
|
255
|
-
inputResponses: {
|
|
260
|
+
inputResponses: { format: { action: 'decline' } },
|
|
256
261
|
});
|
|
257
|
-
await expect(
|
|
262
|
+
await expect(exportReport.handler(input, ctx)).rejects.toThrow(McpError);
|
|
258
263
|
});
|
|
259
264
|
```
|
|
260
265
|
|
|
266
|
+
Seeding an answer this way is right for a handler that treats it as input. A consent gate does not: it acts only on a record it stored when it asked, so an answer seeded alone makes it ask again — see below.
|
|
267
|
+
|
|
261
268
|
`ctx.inputs.dropped` is always `[]` on a mock context — the drop only happens in the SDK's wire decoding, so cover it in an integration test rather than a unit one.
|
|
262
269
|
|
|
270
|
+
Seed `clientCapabilities` to test what a client without a capability gets: a pre-answered `inputResponses` the declared capabilities do not cover — a kind never declared, a form answer (one carrying `content`) from a client that declared only `elicitation.url`, a tool-use sampling answer without `sampling.tools` — never reaches `ctx.inputs`, so the handler asks again. A handler that falls through when a capability is missing (`if (ctx.clientCapabilities?.roots) … else …`) is tested by seeding both shapes.
|
|
271
|
+
|
|
272
|
+
A consent gate that redeems a `ctx.state` record (see `api-context` § *Consent gates*) needs the record in the second round's storage, and each mock context has its own. Copy what round one stored into the round-two context. The record carries the caller, so seed the same `auth` (or none) on both:
|
|
273
|
+
|
|
274
|
+
```ts
|
|
275
|
+
const accept = { confirm: { action: 'accept', content: { confirm: true } } };
|
|
276
|
+
|
|
277
|
+
const first = createMockContext();
|
|
278
|
+
const asked = await expectInputRequired(() => deletePath.handler(input, first));
|
|
279
|
+
const record = await first.state.get(`consent/${asked.requestState}`);
|
|
280
|
+
|
|
281
|
+
const second = createMockContext({ inputResponses: accept, requestState: asked.requestState });
|
|
282
|
+
await second.state.set(`consent/${asked.requestState}`, record);
|
|
283
|
+
await expect(deletePath.handler(input, second)).resolves.toEqual({ deleted: input.path });
|
|
284
|
+
|
|
285
|
+
// The record is spent: the same state again asks for a fresh confirmation.
|
|
286
|
+
await expect(deletePath.handler(input, second)).rejects.toSatisfy(isInputRequiredSignal);
|
|
287
|
+
|
|
288
|
+
// An answer with no record behind it asks as well — nothing was asked, so nothing was confirmed.
|
|
289
|
+
const unasked = createMockContext({ inputResponses: accept });
|
|
290
|
+
await expect(deletePath.handler(input, unasked)).rejects.toSatisfy(isInputRequiredSignal);
|
|
291
|
+
```
|
|
292
|
+
|
|
293
|
+
A record the round-two context holds under another `auth`, or one written by another tool (its `operation` differs), asks again the same way — cover whichever of those the handler's binding is meant to catch.
|
|
294
|
+
|
|
263
295
|
### Mock logger
|
|
264
296
|
|
|
265
297
|
`ctx.log` captures all log calls for inspection. Import `MockContextLogger` from `@cyanheads/mcp-ts-core/testing` and cast `ctx.log` to access the `.calls` array (the cast is necessary because `createMockContext` returns `Context`, which types `log` as `ContextLogger`):
|
|
@@ -4,7 +4,7 @@ description: >
|
|
|
4
4
|
API reference for all utilities exported from `@cyanheads/mcp-ts-core/utils`. Use when looking up utility method signatures, options, peer dependencies, or usage patterns.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "2.
|
|
7
|
+
version: "2.14"
|
|
8
8
|
audience: external
|
|
9
9
|
type: reference
|
|
10
10
|
---
|
|
@@ -37,7 +37,7 @@ Utility exports from `@cyanheads/mcp-ts-core/utils`. Utilities with complex APIs
|
|
|
37
37
|
| `deadlineMs` | `RetryOptions` field | One wall-clock budget across every attempt, backoff, and honored `Retry-After` — the bound `maxRetries` plus a per-attempt timeout cannot express. Four 30s attempts outlast a client's 60s request timeout, so the caller gets a transport timeout instead of the server's classified error. **Thread `attempt.signal` into the attempt's I/O** (`fetchWithTimeout(url, Math.min(30_000, remainingMs), ctx, { signal })`) or the deadline overshoots by one in-flight request. Clock is `AbortController` + `setTimeout` (never `AbortSignal.timeout()`, per the Bun realm mismatch), cleared on return — no timer outlives the call. Expiry rejects with `Timeout` (-32004) carrying `data: { reason: 'retry_deadline_exceeded', deadlineMs, elapsedMs, retryAttempts }` and the last attempt's error as `cause`; **one shape for every expiry**, including the per-attempt `Timeout` (`errorSource: 'FetchSignalTimeout'`) the clock's abort raises inside `fetchWithTimeout` and the raw abort reason a mid-backoff expiry would otherwise surface. No `retryable` flag (a narrower call can still succeed) and no `attempt` index (`retryAttempts` carries it). A backoff that would outlast the remaining budget fails fast with the expiry instead of sleeping into a certain timeout; an honored `Retry-After` that would outlast it takes the `maxDelayMs` exit instead — the attempt's error unchanged, `data.retryAfter` intact, since "wait the window the upstream named" is still the caller's action. **Three clocks stay distinct:** a caller abort on `options.signal` keeps precedence — mid-attempt it rethrows the attempt's error unchanged, mid-backoff it rejects with `signal.reason` itself (an `AbortError` `DOMException` for a reason-less `abort()`), and the handler factory reports either as `RequestCancelled` when the request signal is the one that fired — a single attempt's timeout is `Timeout` with `errorSource: 'FetchTimeout'` and no `reason`, and the expiry is `Timeout` with the `reason`. Unset, behavior is identical to before — attempt counts, delays, log lines, and the exhausted-error shape untouched. Bounds **one** ladder: a tool making three upstream calls threads its own remaining budget into each. |
|
|
38
38
|
| `defaultIsTransient` | `(error: unknown) -> boolean` | The predicate `withRetry` uses when `isTransient` is omitted: an `McpError` with a transient code (`ServiceUnavailable`, `Timeout`, `RateLimited`) unless it carries `data.retryable === false`, `data.reason === 'pacer_shed'`, or `data.errorSource === 'FetchSignalTimeout'` (a caller-side deadline that already fired); any non-`McpError` throw is assumed transient. Exported so `isTransient` — which **replaces** the default outright — can compose instead of mirroring the transient set, which drifts silently when the framework's classification changes: `isTransient: (error) => !isMyBudgetRefusal(error) && defaultIsTransient(error)`, or the inverse `defaultIsTransient(error) \|\| isMyRetryableShape(error)`. The transient code set itself stays private (a module-level `Set` an exported binding could be mutated into framework-wide retry behavior). |
|
|
39
39
|
| `httpErrorFromResponse` | `(response: Response, options?: HttpErrorFromResponseOptions) -> Promise<McpError>` | Maps an HTTP `Response` to a properly classified `McpError` — full status table including 401/403/408/422/429/5xx, body capture (truncated), `retry-after` header, optional `cause`. `error.data` carries `status`/`body` plus the legacy `statusCode`/`responseBody` aliases (identical values), so a consumer can classify either helper's error without knowing which raised it. Use this instead of hand-rolling `if (status === 429) ...` ladders. Reads the response body — `clone()` first if you need it elsewhere. **`error.data` is client-facing** — the framework forwards it verbatim as `structuredContent.error.data` — so the full upstream URL is **omitted by default**: a request URL routinely carries user input, internal identifiers, or an API key in its query string. `includeUrl: true` opts into `data.url` carrying the full `response.url`; with an empty `response.url` no key is added either way, and the message still names the host. Response headers are opt-in on the same footing: `errorHeaders: ['x-ratelimit-remaining-usd', 'x-request-id']` copies the named headers onto `data.headers` under **lowercase** keys — selection is case-insensitive and entries differing only in case collapse to one key, presence follows `Headers.has()` (an empty value is captured as `''`, an absent header adds no key), and a multi-valued field is captured comma-joined as `Headers.get()` returns it. Omitted, empty, or matching nothing, no `headers` key is emitted. `set-cookie` is **never** captured whatever the selector says: it is credential-bearing and `Headers.get()` joins its values into a string that is not a valid reconstruction. Every selected value reaches the client, so never name a header that carries a credential — and a selected `Location` can itself carry a sensitive path, query, or token. `HttpErrorFromResponseOptions`: `service?` (logical name in message, e.g. `'NCBI'`), `captureBody?` (default `true`), `bodyLimit?` (default `500`), `includeUrl?` (default `false`), `errorHeaders?` (default none), `data?` (extra fields merged into `error.data`, overriding defaults on key collision — a caller's own `url` or `headers` still reaches the wire), `cause?`, `codeOverride?` (per-status mapping override). Pairs naturally with `withRetry` — both classify codes the same way. A 501 also carries `data.retryable: false`, so retry fails it fast instead of re-asking for a method the upstream does not implement. |
|
|
40
|
-
| `createPacer` | `(options: PacerOptions) -> Pacer` | FIFO queue in front of one rate-limited upstream — the outbound counterpart to `RateLimiter` (`utils/security`), which is inbound, per-caller, and reject-only, so it cannot queue work against an upstream budget. `pacer.run(task, { signal?, maxWaitMs? })` holds `task` until every `limits` window, `minStartGapMs`, `maxConcurrent`, and the cooldown gate allow it, then calls it with the caller's signal. `PacerOptions`: `name` (author-set telemetry label), `limits` (`{ requests, perMs }[]` — each a sliding window over recorded **start** times, so a slow response never widens the rate the upstream sees; all must allow a start), `minStartGapMs` (**not** expressible through `limits`: `{ requests: 10, perMs: 1000 }` permits ten starts in the same millisecond), `maxConcurrent`, `maxQueueDepth` (absolute backpressure for callers passing no `maxWaitMs`; rejects without arming a timer), `cooldown` (`{ baseMs, maxMs }`). **Shed:** `maxWaitMs` bounds queue time only, never the task. The projected wait is exact over the windows and the gap but a lower bound once `maxConcurrent` binds (a slot frees on an unknowable completion), so enqueue rejects only when that lower bound already exceeds `maxWaitMs` — no false sheds — and a still-queued entry rejects when `maxWaitMs` elapses. The shed error is `rateLimited` (-32003) with `data: { reason: 'pacer_shed', retryAfter, queueDepth }` and **
|
|
40
|
+
| `createPacer` | `(options: PacerOptions) -> Pacer` | FIFO queue in front of one rate-limited upstream — the outbound counterpart to `RateLimiter` (`utils/security`), which is inbound, per-caller, and reject-only, so it cannot queue work against an upstream budget. `pacer.run(task, { signal?, maxWaitMs? })` holds `task` until every `limits` window, `minStartGapMs`, `maxConcurrent`, and the cooldown gate allow it, then calls it with the caller's signal. `PacerOptions`: `name` (author-set telemetry label), `limits` (`{ requests, perMs }[]` — each a sliding window over recorded **start** times, so a slow response never widens the rate the upstream sees; all must allow a start), `minStartGapMs` (**not** expressible through `limits`: `{ requests: 10, perMs: 1000 }` permits ten starts in the same millisecond), `maxConcurrent`, `maxQueueDepth` (absolute backpressure for callers passing no `maxWaitMs`; rejects without arming a timer; bounds **waiters only** — an arrival whose slot is open that instant starts without queueing, so `0` means "run when a slot is free, never wait"), `cooldown` (`{ baseMs, maxMs }`). **Shed:** `maxWaitMs` bounds queue time only, never the task. The projected wait is exact over the windows and the gap but a lower bound once `maxConcurrent` binds (a slot frees on an unknowable completion), so enqueue rejects only when that lower bound already exceeds `maxWaitMs` — no false sheds — and a still-queued entry rejects when `maxWaitMs` elapses, unless its slot opens that same instant. The shed error is `rateLimited` (-32003) with `data: { reason: 'pacer_shed', shedKind, retryAfter, queueDepth }`. `shedKind` (`PacerShedKind`) is `queue_full` (the call would wait behind `maxQueueDepth` waiters), `wait_projected` (the enqueue projection exceeds `maxWaitMs`), or `wait_elapsed` (`maxWaitMs` ran out while queued), and the message follows the kind — a `queue_full` shed names the full queue, not a wait budget. `retryAfter` is seconds until a caller joining behind every remaining waiter could start; while `maxConcurrent` is saturated — a release the projection cannot see — it is floored at the longest wait of any queued caller, the shed one included, minimum 1. `queueDepth` is the waiters still queued. **No `retryable: false`** — to the calling agent a shed is an ordinary rate limit (wait `retryAfter`, call again) and that flag would say the opposite; `defaultIsTransient` reads the `reason` instead, so an enclosing `withRetry` fails fast rather than sleeping past the deadline the shed enforces. **Cooldown gate:** a `RateLimited` thrown by the task closes the gate for every queued caller until an absolute instant, `min(max(baseMs · 2^(consecutive−1), retryAfter), maxMs)` — `maxMs` caps both the doubling and an honored `Retry-After`, so a pathological upstream value cannot park the queue. Absent or unparseable `retryAfter` leaves the doubling; any other error leaves the gate open and the count untouched, and a shed (`reason: 'pacer_shed'`) from a pacer nested inside the task is local backpressure, never a gate closure. The first success resets the count, and so does a gate that has stood open for `maxMs`: the next rate limit starts over at `baseMs`, while one arriving sooner — the gate still closed included — keeps doubling, so continuous demand under a sustained limit keeps its capped backoff. **`pacer.cooldown`** samples the gate as `PacerCooldownState` `{ remainingMs, consecutive }`: `remainingMs` is the shared gate, not one rate limit's own computation (rate limits landing together close one gate at the later instant), so a task's rejection handler can report it on the server's own error — the pacer never writes to the task's error. Both stay 0 without `cooldown`. **Composition:** `withRetry(({ signal }) => pacer.run(fn, { signal }), { signal, deadlineMs })` — retry outside, pacer inside, so each attempt re-queues and is re-paced. Because the gate is an absolute instant rather than a duration counted from dequeue, retry's `Retry-After` sleep and the gate overlap in wall-clock instead of summing: the window is waited once, not twice. **Lifecycle:** timers and `AbortSignal` only, process-local; the dispatch timer is `unref()`'d where supported; `dispose()` / `[Symbol.dispose]()` clears it and rejects queued waiters with `RequestCancelled` (in-flight tasks are left to finish) — wire it through `createApp({ teardown })`. On Workers state is per-isolate so the limits bind per isolate, OTel is off so the metrics are inert, and `createWorkerHandler` accepts no `teardown`. Metrics: `mcp.pacer.queue_depth`, `mcp.pacer.wait`, `mcp.pacer.sheds`, `mcp.pacer.cooldowns`, attributed by `mcp.pacer.name` only — see `api-telemetry`. |
|
|
41
41
|
| `httpStatusToErrorCode` | `(status: number) -> JsonRpcErrorCode \| undefined` | Sync status → code lookup. Returns `undefined` for 1xx/2xx. A 3xx maps to `InvalidRequest` — it reaches error mapping under `redirect: 'manual'`, where the request as sent cannot be served at this URL, and that code is outside `withRetry`'s transient set since re-issuing returns the same redirect. Use when you need just the code without a `Response` object handy. No status maps to `InternalError` — that code means *this* server failed, which a remote status cannot establish; every 5xx is `ServiceUnavailable` (or `Timeout` for 504) and so picks up `withRetry`'s default transient policy. |
|
|
42
42
|
|
|
43
43
|
---
|
|
@@ -4,7 +4,7 @@ description: >
|
|
|
4
4
|
Cloudflare Workers deployment using `createWorkerHandler` from `@cyanheads/mcp-ts-core/worker`. Covers the full handler signature, binding types, CloudflareBindings extensibility, runtime compatibility guards, and wrangler.toml requirements.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "1.
|
|
7
|
+
version: "1.9"
|
|
8
8
|
audience: external
|
|
9
9
|
type: reference
|
|
10
10
|
---
|
|
@@ -204,6 +204,8 @@ export function getServerConfig() {
|
|
|
204
204
|
|
|
205
205
|
**`in-memory` storage is volatile.** Data stored with the `in-memory` provider is lost between cold starts and is not shared across Worker instances. Use `cloudflare-kv`, `cloudflare-r2`, or `cloudflare-d1` for any state that must persist or be shared.
|
|
206
206
|
|
|
207
|
+
**Multi-round-trip retries land on any isolate.** Each 2026-07-28 request builds its own `McpServer`, and the retry after an `input_required` result can be routed to a different isolate. Set `MCP_REQUEST_STATE_KEY` (≥ 32 bytes) as a Worker secret — it is a core binding, injected like the rest — so every isolate seals and verifies `requestState` with the same key, and keep a consent gate's record (`api-context` § *Consent gates*) in `cloudflare-d1`. `in-memory` loses it to the next isolate, so the retry asks again. Never `cloudflare-kv`: it is eventually consistent, so a retry served elsewhere may not see the record yet, and a deletion may not yet stop an immediate replay — it widens the window concurrent retries already have, since redeeming is not atomic on any provider until `ctx.state` gains a `take` ([#593](https://github.com/cyanheads/mcp-ts-core/issues/593)).
|
|
208
|
+
|
|
207
209
|
**Node-only utilities throw in Workers.** `scheduler` (`node-cron`), `sanitizePath` (fs-based), and `filesystem` storage provider all throw `ConfigurationError` when called from a Worker. Guard with `runtimeCaps.isNode` or avoid entirely.
|
|
208
210
|
|
|
209
211
|
**DataCanvas is unavailable in Workers.** DuckDB has no V8-isolate build, so `core.canvas` is always `undefined` on Workers. Setting `CANVAS_PROVIDER_TYPE=duckdb` (the only non-default value) in `wrangler.toml` triggers a fail-closed `ConfigurationError` at init time:
|
|
@@ -4,7 +4,7 @@ description: >
|
|
|
4
4
|
Design the tool surface, resources, and service layer for a new MCP server. Use when starting a new server, planning a major feature expansion, or when the user describes a domain/API they want to expose via MCP. Produces a design doc at docs/design.md that drives implementation.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "2.
|
|
7
|
+
version: "2.31"
|
|
8
8
|
audience: external
|
|
9
9
|
type: workflow
|
|
10
10
|
---
|
|
@@ -263,9 +263,9 @@ A reference tool is the surface's decoder ring: which codes exist, what they mea
|
|
|
263
263
|
|
|
264
264
|
Tools that perform multi-step mutations (the Workflow shape) have two safety considerations beyond single-call tools. Both are about giving the agent — and the human behind it — a chance to catch a bad invocation before it commits.
|
|
265
265
|
|
|
266
|
-
**Confirmation-gated destructive modes, with an annotation fallback.** When a workflow's `mode` parameter switches between safe and destructive arms (`draft` vs `send`, `plan` vs `apply`), gate the destructive arm on a confirmation the handler asks for via `ctx.requestInput(...)`, so a human approves before the irreversible step fires. The handler is re-entered with the answer on `ctx.inputs`; it does not `await` mid-call.
|
|
266
|
+
**Confirmation-gated destructive modes, with an annotation fallback.** When a workflow's `mode` parameter switches between safe and destructive arms (`draft` vs `send`, `plan` vs `apply`), gate the destructive arm on a confirmation the handler asks for via `ctx.requestInput(...)`, so a human approves before the irreversible step fires. The handler is re-entered with the answer on `ctx.inputs`; it does not `await` mid-call. The answer alone does not prove the prompt was shown — a client can send one unasked, and any `requestState` replays within its lifetime — so the gate stores the operation, the caller, the confirmed target, and a hash of what it holds in a `ctx.state` record keyed by a random id, sends only that id as `requestState`, and redeems the record before the destructive arm runs (`api-context` § *Consent gates*). Plan shared storage for that record when a 2026-07-28 retry can reach another instance (`filesystem`, `supabase`, or `cloudflare-d1` — not `cloudflare-kv`), and set `MCP_REQUEST_STATE_KEY` so a retry can only carry state the server minted. Redeeming is not atomic — concurrent retries on one id can each pass the gate until `ctx.state` gains an atomic `take` ([#593](https://github.com/cyanheads/mcp-ts-core/issues/593)) — so an arm that must not run twice is idempotent per record.
|
|
267
267
|
|
|
268
|
-
The gate is always *reachable* — `ctx.requestInput` is present on every transport and both protocol revisions (2025-11-25 legacy, 2026-07-28 current) — but it is not always *answerable*: a client that never fulfils the `input_required` result simply doesn't retry, and the destructive step never runs. The same holds for a 2025-11-25 HTTP client when the server runs `MCP_SESSION_MODE=stateless`: the legacy round-trip shim still runs, but its capability gate refuses because the serving instance never processed `initialize` — the destructive step never fires. That is the safe outcome, but it makes the tool unusable for those clients, so a server built around such a gate declares `createApp({ sessionMode: { default: 'stateful', require: 'stateful' } })` and refuses to start stateless rather than degrading (`api-context` § `ctx.requestInput`). Keep `destructiveHint: true` in annotations so those clients' own approval flows still surface the risk. A decline is terminal — the handler fails the call rather than re-asking, which would loop until the round budget runs out.
|
|
268
|
+
The gate is always *reachable* — `ctx.requestInput` is present on every transport and both protocol revisions (2025-11-25 legacy, 2026-07-28 current) — but it is not always *answerable*: a client that never fulfils the `input_required` result simply doesn't retry, and — with the record pattern, which proceeds only on a record it redeemed — the destructive step never runs. The same holds for a 2025-11-25 HTTP client when the server runs `MCP_SESSION_MODE=stateless`: the legacy round-trip shim still runs, but its capability gate refuses because the serving instance never processed `initialize` — the destructive step never fires. That is the safe outcome, but it makes the tool unusable for those clients, so a server built around such a gate declares `createApp({ sessionMode: { default: 'stateful', require: 'stateful' } })` and refuses to start stateless rather than degrading (`api-context` § `ctx.requestInput`). Keep `destructiveHint: true` in annotations so those clients' own approval flows still surface the risk. A decline is terminal — the handler fails the call rather than re-asking, which would loop until the round budget runs out. `ctx.clientCapabilities` never decides whether to ask: a gate that skips its prompt when `elicitation` is undeclared is the bypass the gate exists to prevent.
|
|
269
269
|
|
|
270
270
|
**Safe defaults on parameters that determine blast radius.** When a workflow accepts a parameter that controls how far-reaching a mutation is, default to the safer value. A bulk file-update tool defaulting `mode: 'preview'` (no writes) means a sloppy agent call shows a diff rather than blasting changes; an apply-plan tool defaulting `dryRun: true` means a misread plan previews rather than executes; an object-delete tool requiring an explicit `confirmCount` matching the result-set size means an unscoped query can't silently nuke a million rows. Agents that genuinely want the destructive behavior have to name it explicitly, which surfaces intent in the tool call and in logs.
|
|
271
271
|
|
|
@@ -332,7 +332,7 @@ nctIds: z.union([z.string(), z.array(z.string()).max(5)])
|
|
|
332
332
|
| Delimiter-joined list where an array is accepted | `"US,JP,KR"` → `["US","JP","KR"]` | Split on the documented separator |
|
|
333
333
|
| Spelled-out vs. abbreviated name | `"Houston, Texas"` → `"Houston, TX"` | Normalize against the bundled name table |
|
|
334
334
|
|
|
335
|
-
These are **value**-level, and the mappings are domain knowledge — settle them per input in the design doc's param table. Argument **key** names are not: the framework
|
|
335
|
+
These are **value**-level, and the mappings are domain knowledge — settle them per input in the design doc's param table. Argument **key** names are not: the framework rewrites declared and case-style key aliases and drops client-added root keys before the schema sees the arguments, and repairs a JSON-stringified array or object, or an integer sent for a string, against the tool's own schema after a failed parse — so an ID field stays `z.string()`, never a `string | number` union. Don't re-implement any of that per server — see `add-tool` § *Three things the framework fixes before the schema sees the arguments*.
|
|
336
336
|
|
|
337
337
|
This resolves one submitted value to one canonical value, and does not loosen the strict token match in [MCP-side list filtering](#mcp-side-list-filtering), which scores a query against many candidate names.
|
|
338
338
|
|
|
@@ -423,7 +423,7 @@ Two params, two behaviors — keep them named distinctly:
|
|
|
423
423
|
|
|
424
424
|
Errors are part of the tool's interface — design them during the design phase, not as an afterthought. Three aspects: **the contract** (which failures are public), **classification** (what error code), and **messaging** (what the LLM reads).
|
|
425
425
|
|
|
426
|
-
**Declare a typed contract for domain failures.** When a tool has known failure modes the agent should plan around (`no_match`, `queue_full`, `vendor_down`), enumerate them as `errors: [{ reason, code, when, recovery, retryable? }]` on the definition. `recovery` is required metadata — the agent's next move when this failure fires (≥ 5 words, lint-validated;
|
|
426
|
+
**Declare a typed contract for domain failures.** When a tool has known failure modes the agent should plan around (`no_match`, `queue_full`, `vendor_down`), enumerate them as `errors: [{ reason, code, when, recovery, retryable? }]` on the definition. `recovery` is required metadata — the agent's next move when this failure fires (≥ 5 words, lint-validated; the framework sends it on the wire as `data.recovery.hint` with any failure carrying that reason and no hint of its own). The framework types `ctx.fail(reason, …)` against the declared reason union (typos become TS errors) and auto-populates `data.reason` on the thrown error for stable observability. The error reaches clients with parity across both surfaces — `structuredContent.error` (Claude Code) and `content[]` text (Claude Desktop). Baseline codes (`InternalError`, `ServiceUnavailable`, `Timeout`, `ValidationError`, `SerializationError`, `RequestCancelled`) bubble from anywhere and don't need to be enumerated. Mark an entry the service layer throws, rather than the handler, with `thrownBy: 'service'` so the conformance lint doesn't report it as a reason the handler never raises. See `api-errors` skill for the full pattern.
|
|
427
427
|
|
|
428
428
|
**Classify errors by origin.** Different error sources need different codes and different recovery guidance. Map the failure modes for each tool during design:
|
|
429
429
|
|
|
@@ -702,7 +702,7 @@ Items without an `If …:` prefix apply to every design. Conditional items only
|
|
|
702
702
|
- [ ] **If the server has workflow tools:** call-flow documented (upstream sequence + mode arms) in design doc's Workflow Analysis
|
|
703
703
|
- [ ] **If state-aware procedural guidance adds value:** instruction tool considered with `nextToolSuggestions` pre-filled from diagnostics
|
|
704
704
|
- [ ] **If any tool is config-gated:** nothing routes to it while the gate is off — recovery strings, notices, and `guidance` name a callable target or state the capability is unavailable, and structured follow-ups naming it are emitted only under the config that registers it
|
|
705
|
-
- [ ] **If workflow tools have destructive modes:** destructive arm gated on a `ctx.requestInput` confirmation
|
|
705
|
+
- [ ] **If workflow tools have destructive modes:** destructive arm gated on a `ctx.requestInput` confirmation that redeems a `ctx.state` consent record bound to the operation, caller, and target (shared storage when retries can reach another instance, `MCP_REQUEST_STATE_KEY` set, an arm that must not run twice idempotent per record), with `destructiveHint` annotation so clients that never fulfil the round still surface the risk
|
|
706
706
|
- [ ] **If any tool calls `ctx.requestInput`:** `createApp()` declares `sessionMode` with `require: 'stateful'`
|
|
707
707
|
- [ ] **If any output carries text other people wrote:** those fields listed, `format()` quotes or fences free text and flattens CR/LF in inline slots, and the server instructions say the content is data
|
|
708
708
|
- [ ] **If a parameter determines blast radius:** safe default set (e.g., `mode: 'preview'`, `dryRun: true`, `confirmCount` required)
|
|
@@ -4,7 +4,7 @@ description: >
|
|
|
4
4
|
Exercise tools, resources, and prompts against a live HTTP server via MCP JSON-RPC over curl. Starts the server, surfaces the catalog, runs real and adversarial inputs, measures every call (bytes, token estimate, wall-clock) and weighs the catalog, and produces a tight report with concrete findings and numbered follow-up options. Use after adding or modifying definitions, or when the user asks to test, try out, or verify their MCP surface.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "2.
|
|
7
|
+
version: "2.18"
|
|
8
8
|
audience: external
|
|
9
9
|
type: debug
|
|
10
10
|
---
|
|
@@ -19,7 +19,7 @@ Unit tests (`add-test` skill) verify handler logic with mocked context. Field te
|
|
|
19
19
|
|
|
20
20
|
This skill drives an HTTP server because curl + JSON-RPC is the most reliable harness for shell-based agents. The same handlers run on both transports — only the framing differs — so HTTP exercises the full functional surface. Both HTTP session modes are covered: a durable `Mcp-Session-Id` session, and the sessionless initialization a `MCP_SESSION_MODE=stateless` server performs.
|
|
21
21
|
|
|
22
|
-
**Stdio coverage is a boot check only — run this before Step 1.** Run `bun run rebuild && bun run start:stdio < /dev/null`, and confirm the startup logs look clean (
|
|
22
|
+
**Stdio coverage is a boot check only — run this before Step 1.** Run `bun run rebuild && bun run start:stdio < /dev/null`, and confirm the startup logs look clean: the `Core services constructed — N tool(s) …` record lists every registered tool, resource, and prompt in its `tools` / `resources` / `prompts` fields — the message text shows only counts — and a definition missing from them was never passed to `createApp()`. No errors/warnings, no missing-config gripes. The emoji startup banner prints only to a terminal, so its absence from an agent's shell is not a finding. Redirecting stdin is what ends the run: the server treats EOF as a shutdown signal, boots fully, then exits on its own, so the log also shows the graceful-shutdown path. Do not background it and reach for `pkill` — a pattern like `pkill -f dist/index.js` matches every other stdio MCP server on the machine, including the ones the calling agent's own session is connected to. Pino logs go to stderr in stdio mode (stdout is reserved for JSON-RPC), so they print straight to the terminal when you run interactively. No need to call tools over stdio — the HTTP pass already covered handler behavior.
|
|
23
23
|
|
|
24
24
|
---
|
|
25
25
|
|
|
@@ -402,7 +402,7 @@ Treat any hit as a `ux` finding in the report. The authoring rule lives under *T
|
|
|
402
402
|
|:------------------------------------------------|:-------------|
|
|
403
403
|
| `include` / `fields` / `expand` / `view` / `projection` parameter | Field selection: non-default value renders requested fields |
|
|
404
404
|
| Array return with `query` / `filter` inputs | Empty result: does response explain *why* (echo criteria, suggest broadening)? |
|
|
405
|
-
| Identifier, code, or enum-ish input (an ID format, a classification code, a unit, a place name, a list the docs say may be comma-joined) | Value-variant tolerance: re-send the happy-path call with each obvious variant of that value — lowercase, the bare leaf of a hierarchical code, a common domain alias, a delimiter-joined list where an array is accepted, the spelled-out form of an abbreviated name. Pass is either outcome: the call succeeds, or it fails with an error naming the expected shape. A miss or a bare validation failure on a variant that maps one-to-one onto a valid value is a `ux` finding. Probe **values** — variants of the argument *key* name, and a JSON-stringified array as a value, are handled by the framework, not the server. |
|
|
405
|
+
| Identifier, code, or enum-ish input (an ID format, a classification code, a unit, a place name, a list the docs say may be comma-joined) | Value-variant tolerance: re-send the happy-path call with each obvious variant of that value — lowercase, the bare leaf of a hierarchical code, a common domain alias, a delimiter-joined list where an array is accepted, the spelled-out form of an abbreviated name. Pass is either outcome: the call succeeds, or it fails with an error naming the expected shape. A miss or a bare validation failure on a variant that maps one-to-one onto a valid value is a `ux` finding. Probe **values** — variants of the argument *key* name, and a JSON-stringified array or object or an integer sent for a string as a value, are handled by the framework, not the server. |
|
|
406
406
|
| Batch / bulk input (arrays of IDs, multi-item ops) | Partial success: mix valid + invalid items |
|
|
407
407
|
| `annotations.readOnlyHint: true` | Confirm no mutation happened |
|
|
408
408
|
| `annotations.idempotentHint: true` | Call twice with same input — safe? |
|
|
@@ -436,7 +436,7 @@ When a call surprises you — slow, hangs, returns terse output, surfaces an unh
|
|
|
436
436
|
|
|
437
437
|
- **`content[]` is an array of blocks — read all of them, never just `content[0]`.** A success result is assembled as `[...ctx.content media blocks, ...the format()/JSON domain render, ...the enrichment trailer]`. Everything the handler put on `ctx.enrich` — empty-result notices, totals, query echoes, truncation disclosure — renders in that trailer, a **separate trailing block**, not inside the `format()` block. Quoting `content[0].text` and reporting those fields as absent from `content[]` is a false parity gap; the suggested fix (render them in `format()` too) would double-render them. Dump `.result.content` in full before claiming drift.
|
|
438
438
|
- Tool domain errors return `{result: {content: [...], isError: true}}` — they live in `result`, not `error`. Check `isError`, not the JSON-RPC error field.
|
|
439
|
-
- **Tool error code/reason** rides on `result.structuredContent.error.{code, message, data?.reason}` — inspect that, not just the text. `data`
|
|
439
|
+
- **Tool error code/reason** rides on `result.structuredContent.error.{code, message, data?.reason}` — inspect that, not just the text. `data` carries what the handler threw as an `McpError` (or a `ZodError`'s issues) plus the framework's `data.requestId`; plain `throw new Error(...)` won't populate `data.reason`. Use `ctx.fail`-thrown errors when the contract reason matters — a declared reason arrives with its contract `recovery` as `data.recovery.hint` even when the throw site passed none. The text in `result.content[0].text` mirrors the message, adds `Recovery: <hint>` when `data.recovery.hint` says something the message does not already say, and closes with `(reason <reason> · not retryable · request <id>)` for whichever of `data.reason` / `data.retryable` / `data.requestId` is present — the numeric code stays JSON-only. The request id matches the `requestId` on that call's server log records.
|
|
440
440
|
- **Resource errors** are JSON-RPC-level — they appear in the top-level `error.{code, data.reason}` field, not inside `result`. Resource handlers re-throw rather than producing an `isError` envelope.
|
|
441
441
|
- JSON-RPC `error` only appears for protocol issues (bad session, malformed envelope, unknown method).
|
|
442
442
|
- `mcp_call` already strips SSE framing. Pipe to `jq` for readability.
|
|
@@ -500,7 +500,7 @@ End with:
|
|
|
500
500
|
|
|
501
501
|
## Checklist
|
|
502
502
|
|
|
503
|
-
- [ ] Stdio boot check completed — `bun run rebuild && bun run start:stdio < /dev/null` shows clean startup (
|
|
503
|
+
- [ ] Stdio boot check completed — `bun run rebuild && bun run start:stdio < /dev/null` shows clean startup (every expected definition listed in the `Core services constructed` record's `tools` / `resources` / `prompts` fields, no errors) and a graceful shutdown on EOF
|
|
504
504
|
- [ ] HTTP server built and started; real port parsed from log
|
|
505
505
|
- [ ] Session initialized (a stateless server returns an empty `sid` — still a pass); `notifications/initialized` sent; negotiated protocol version matches the requested one (a downgrade is a finding)
|
|
506
506
|
- [ ] Catalog surfaced and presented; descriptions audited for leaks (implementation details, meta-coaching, consumer-aware phrasing)
|
|
@@ -4,7 +4,7 @@ description: >
|
|
|
4
4
|
Land working-tree changes as logical commits — the work grouped by concern, topped by a release commit (version bump, changelog, regenerated artifacts). The work commits land first, then the version bump, verification, and the release commit on top. Stops at "committed locally on main" — or, when the project releases through a release PR, at "release branch pushed, PR open". No tag, no push to main, no publish: the release-and-publish skill merges, tags, and ships from here. Distilled from the git_wrapup_instructions protocol.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "1.
|
|
7
|
+
version: "1.27"
|
|
8
8
|
audience: external
|
|
9
9
|
type: workflow
|
|
10
10
|
---
|
|
@@ -144,8 +144,8 @@ When every concern is committed, `git status` is clean. That clean tree is what
|
|
|
144
144
|
Every file that declares a version must be updated. Skip any file that doesn't exist in the project. For `@cyanheads/mcp-ts-core` projects:
|
|
145
145
|
|
|
146
146
|
- `package.json` — `version`
|
|
147
|
-
- `server.json` — top-level `version` AND every `packages[].version` entry
|
|
148
|
-
- `manifest.json` (if present) — `version`.
|
|
147
|
+
- `server.json` — top-level `version` AND every `packages[].version` entry. `lint:mcp` flags a mismatch at either level
|
|
148
|
+
- `manifest.json` (if present) — `version`. Packaging validation fails on a mismatch, and on a scoped `name` (use `bls-mcp-server`, not `@cyanheads/bls-mcp-server`)
|
|
149
149
|
- `.claude-plugin/plugin.json` and `.codex-plugin/plugin.json` (if present) — `version`. Packaging validation fails on a mismatch; `.codex-plugin/mcp.json` is connection config and carries none
|
|
150
150
|
- `README.md` — version badge. Packaging validation fails on a mismatch with `package.json`; a literal `-` in a prerelease is escaped as `--` (`Version-0.14.0--rc.1-`)
|
|
151
151
|
- `CLAUDE.md` / `AGENTS.md` — if they pin a version string
|
|
@@ -194,15 +194,16 @@ Both scripts are idempotent — safe to run even if nothing changed.
|
|
|
194
194
|
|
|
195
195
|
### 7. Run the verification gate
|
|
196
196
|
|
|
197
|
-
The stack being shipped must pass verification.
|
|
197
|
+
The stack being shipped must pass verification. All must succeed:
|
|
198
198
|
|
|
199
199
|
```bash
|
|
200
200
|
bun run devcheck
|
|
201
|
+
bun run rebuild
|
|
201
202
|
bun run test:all # or `bun run test` if no test:all script exists
|
|
202
203
|
bun run test:package # only if the script exists — NOT part of test:all
|
|
203
204
|
```
|
|
204
205
|
|
|
205
|
-
**If
|
|
206
|
+
**If any fails, halt.** Do not bypass verification to land the release commit.
|
|
206
207
|
|
|
207
208
|
The work is already committed by this point, so the fix is a new commit on top of the stack, under step 3's conventions — never `git commit --amend`, never a rebase, reset, or any other rewrite of a commit the stack already carries. Land the fix, then re-run this step. The same holds when the gate passes but leaves the tree dirty: `devcheck` auto-fixes as it runs, and a formatter fix to a file committed in step 3 is a follow-up commit of its own, not something to fold into the release commit.
|
|
208
209
|
|
|
@@ -298,13 +299,14 @@ If the working tree isn't clean or the release commit isn't at HEAD, something w
|
|
|
298
299
|
|
|
299
300
|
- [ ] Diff reviewed end-to-end before the first commit
|
|
300
301
|
- [ ] Work concerns committed before the version bump — a version-bearing file a work concern also touches ships whole in that concern's commit, so the release commit brings it the version hunk alone
|
|
301
|
-
- [ ] Version bumped in every declaring file (`package.json`, `server.json`, `manifest.json`, `.claude-plugin/plugin.json`, `.codex-plugin/plugin.json`, README badge, `CLAUDE.md`/`AGENTS.md` if they pin a version) —
|
|
302
|
+
- [ ] Version bumped in every declaring file (`package.json`, `server.json`, `manifest.json`, `.claude-plugin/plugin.json`, `.codex-plugin/plugin.json`, README badge, `CLAUDE.md`/`AGENTS.md` if they pin a version) — `devcheck` flags a mismatch in `server.json` (both levels), `manifest.json`, both plugin manifests, and the README badge; step 4's straggler grep covers the docs and Dockerfile labels
|
|
302
303
|
- [ ] GH issues addressed by this work commented with what landed (if working from GH issues)
|
|
303
304
|
- [ ] Docs updated for any new or changed features
|
|
304
305
|
- [ ] Changelog authored at `changelog/<major.minor>.x/<version>.md`
|
|
305
306
|
- [ ] `CHANGELOG.md` rollup regenerated (`bun run changelog:build`)
|
|
306
307
|
- [ ] `docs/tree.md` regenerated if structure changed (`bun run tree`)
|
|
307
308
|
- [ ] `bun run devcheck` passes
|
|
309
|
+
- [ ] `bun run rebuild` succeeds
|
|
308
310
|
- [ ] `bun run test:all` (or `test`) passes
|
|
309
311
|
- [ ] `bun run test:package` passes, when the project defines it — it guards the public-export manifest and `test:all` does not run it
|
|
310
312
|
- [ ] Release PR mode: stack committed on `release/<version>`, never on `main`
|
|
@@ -4,7 +4,7 @@ description: >
|
|
|
4
4
|
Pick and run a multi-phase workflow that chains foundational task skills (`git-wrapup`, `release-and-publish`, `maintenance`, `field-test`, `setup`, etc.) end-to-end. Routes user intent to a workflow file under `workflows/` — greenfield builds, maintenance + release, field-test + fix, or known-work + release. Single source for the universal rules (no commits without authorization, no destructive git, no marketing language), the orchestrator posture (own the goal, ground sub-agents in primary sources, verify against the goal), and the sub-agent strategy (orient block, parallel fanout, isolation, normalization) that apply across every workflow. Sub-agents are an optional capability — workflows run linearly when fanout isn't available.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "1.
|
|
7
|
+
version: "1.12"
|
|
8
8
|
audience: external
|
|
9
9
|
type: workflow
|
|
10
10
|
---
|
|
@@ -68,8 +68,8 @@ The orchestrator owns the goals. Workflow phases are not "run skill X" — they
|
|
|
68
68
|
|
|
69
69
|
Before running a phase (or spawning a sub-agent for it), write down four things:
|
|
70
70
|
|
|
71
|
-
1. **Goal** — the verifiable end state this phase must produce. Concrete and testable: "v0.5.2 tag exists at HEAD
|
|
72
|
-
2. **Primary sources** — the specific files, GH issues, and reference docs the sub-agent must read directly. Inlining content into the prompt is a paraphrase that loses nuance; agents grounded in the source catch details the orchestrator's summary missed. For GH issues, instruct
|
|
71
|
+
1. **Goal** — the verifiable end state this phase must produce. Concrete and testable: "v0.5.2 tag exists at HEAD and passes `bun run release:github -- --check`; `bun run devcheck` green; `npm view <pkg>@0.5.2` resolves." Not fuzzy: "ran the release-and-publish skill."
|
|
72
|
+
2. **Primary sources** — the specific files, GH issues, and reference docs the sub-agent must read directly. Inlining content into the prompt is a paraphrase that loses nuance; agents grounded in the source catch details the orchestrator's summary missed. For GH issues, instruct the three reads in the Orient block — `gh issue view N` (the body), `gh issue view N --comments` (the thread; without a TTY it prints no body), and the timeline cross-reference query (what references the issue, including cross-repo). The orchestrator reads these sources too (to construct the prompt), but that's prompt construction, not a substitute for the sub-agent reading them.
|
|
73
73
|
3. **Path** — the Tier 1 skill(s) and steps that get to the goal. This is what gets handed to the sub-agent.
|
|
74
74
|
4. **Verification** — the read-only checks that confirm the goal was hit. Defined upfront, not as an afterthought.
|
|
75
75
|
|
|
@@ -79,7 +79,7 @@ Why the framing matters:
|
|
|
79
79
|
- **Sub-agent self-reports describe intent, not always reality.** A goal you wrote down beforehand is the falsification target — the sub-agent's report is a hypothesis to verify against it.
|
|
80
80
|
- **Replanning is local.** When verification fails, the goal is unchanged; the orchestrator picks a different path (re-spawn with the failure context, re-slice the work, intervene directly). Phase rework doesn't cascade.
|
|
81
81
|
|
|
82
|
-
**Inform without inlining.** An enhanced sub-agent prompt names the specific primary sources and the goal — it does NOT paraphrase them. "Review GH issue #123 (read it via `gh issue view 123 --comments`); the goal is X; verify with Y" is the right shape. Pasting the issue body into the prompt forces the sub-agent to work from a paraphrase. Let the sub-agent read the source and explore for additional context as needed.
|
|
82
|
+
**Inform without inlining.** An enhanced sub-agent prompt names the specific primary sources and the goal — it does NOT paraphrase them. "Review GH issue #123 (read it via `gh issue view 123` and `gh issue view 123 --comments`); the goal is X; verify with Y" is the right shape. Pasting the issue body into the prompt forces the sub-agent to work from a paraphrase. Let the sub-agent read the source and explore for additional context as needed.
|
|
83
83
|
|
|
84
84
|
## Sub-Agent Strategy (if available)
|
|
85
85
|
|
|
@@ -122,8 +122,9 @@ order. If any file does not exist, note it and continue.
|
|
|
122
122
|
5. Read the skill file(s) for this task: `[Tier 1 skill paths]`.
|
|
123
123
|
6. Read the primary sources for this task directly — design docs (`docs/design.md`),
|
|
124
124
|
GH issues, handoff documents, reference/gold-standard files. For a GH issue, read
|
|
125
|
-
|
|
126
|
-
- `gh issue view <N
|
|
125
|
+
the body, the comment thread, and its cross-references — three separate reads:
|
|
126
|
+
- `gh issue view <N>` — the body
|
|
127
|
+
- `gh issue view <N> --comments` — the comment thread (without a TTY it prints no body, so it never replaces the read above)
|
|
127
128
|
- `gh api 'repos/{owner}/{repo}/issues/<N>/timeline' --paginate --jq '.[] | select(.event=="cross-referenced") | .source.issue | "\(.repository.full_name)#\(.number) — \(.title)"'` — issues/PRs that reference this one, including from other repos
|
|
128
129
|
List each source explicitly: `[primary source paths and gh commands]`. Skip this
|
|
129
130
|
step only if no primary source applies (rare).
|
|
@@ -4,7 +4,7 @@ description: >
|
|
|
4
4
|
Workflow: field-test one or more existing MCP server projects against the live upstream API, file GH issues for valid findings, deploy fix sub-agents per server, optionally loop until clean, then wrap up and release. Chains the `field-test`, `report-issue-local`, `tool-defs-analysis`, `code-simplifier`, `git-wrapup`, and `release-and-publish` skills. Read `../SKILL.md` first for the universal rules and sub-agent strategy.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "1.
|
|
7
|
+
version: "1.2"
|
|
8
8
|
audience: external
|
|
9
9
|
type: workflow
|
|
10
10
|
---
|
|
@@ -56,8 +56,8 @@ Each phase's Objective column is the goal state per target — the verifiable en
|
|
|
56
56
|
| 3 | Fix | Per target: priority issues fixed in source, tests updated, `devcheck` + `test` green, each issue commented with fix details, working tree dirty for review | parallel fanout (one sub-agent per target — hard constraint) | gate-free |
|
|
57
57
|
| 4 | Verify | Per target: full diff cold-reviewed; simplified if warranted; each fix re-exercised against the running server with actual tool output in the summary | parallel fanout | **barrier** — orchestrator loop decision (human/evidence-based: proceed, loop, or surface to user) |
|
|
58
58
|
| 5 | Loop decision | Orchestrator decision recorded — proceed to release, loop another field-test cycle, or pause/surface to user. Evidence-based | orchestrator (serial) | **barrier** — release authorization required before advancing |
|
|
59
|
-
| 6 | Wrap-up + release | (Optional) Per target: fixes
|
|
60
|
-
| 7 | Issue cleanup | Every GH issue that shipped a fix closed
|
|
59
|
+
| 6 | Wrap-up + release | (Optional) Per target: fixes grouped into one commit per concern (a file never splits across commits) with a release commit on top; annotated tag passing `bun run release:github -- --check` (flat bullets with issue backlinks, changelog link last); published per repo visibility | parallel fanout (Bash git only) | gate-free |
|
|
60
|
+
| 7 | Issue cleanup | Every GH issue that shipped a fix closed (reason: completed) carrying exactly one what-landed comment that cites the version; skipped issues remain open | orchestrator (serial) | — |
|
|
61
61
|
|
|
62
62
|
Phase 6 is optional — stop earlier if release isn't authorized. Phase 7 only runs if Phase 6 ran.
|
|
63
63
|
|
|
@@ -92,7 +92,7 @@ Orchestrator verifies filed issues exist via `gh issue list -R <owner>/<repo>` p
|
|
|
92
92
|
**One sub-agent per target — hard constraint.** No file-locking system exists for concurrent edits; multiple agents touching the same server's `src/` will conflict.
|
|
93
93
|
|
|
94
94
|
Each sub-agent:
|
|
95
|
-
1. Reads all open issues for its target via `gh issue list`
|
|
95
|
+
1. Reads all open issues for its target via `gh issue list`, then `gh issue view N` and `gh issue view N --comments` per issue (body, then thread — the body alone misses clarifications)
|
|
96
96
|
2. **Validates each issue against source code** — a "fixed" issue is a misdiagnosed one if validation fails
|
|
97
97
|
3. Implements fixes in priority order: security → bugs → UX
|
|
98
98
|
4. Rebuilds after each fix or group of related fixes
|
|
@@ -143,19 +143,7 @@ The changelog carries the depth; the tag annotation covers every change at headl
|
|
|
143
143
|
|
|
144
144
|
**Version bump.** Default **patch** for field-test fix releases. **Minor** when enhancements are bundled in.
|
|
145
145
|
|
|
146
|
-
**Tag annotation format
|
|
147
|
-
|
|
148
|
-
```
|
|
149
|
-
Field-test bug fixes across N tools
|
|
150
|
-
|
|
151
|
-
Fixed:
|
|
152
|
-
- <tool_name>: <one-line fix description> (#<issue>)
|
|
153
|
-
- <tool_name>: <one-line fix description> (#<issue>)
|
|
154
|
-
|
|
155
|
-
<test count>; `bun run devcheck` clean.
|
|
156
|
-
```
|
|
157
|
-
|
|
158
|
-
Add a `Security:` section when the changelog frontmatter sets `security: true`.
|
|
146
|
+
**Tag annotation.** Written at release time in the `release-and-publish` step 4 format — a short theme subject, flat bullets with `(#N)` backlinks, the changelog link last; `bun run release:github -- --check` enforces the shape before the push.
|
|
159
147
|
|
|
160
148
|
**Wrap-up scope.** Determined by repo visibility:
|
|
161
149
|
|
|
@@ -167,9 +155,11 @@ Add a `Security:` section when the changelog frontmatter sets `security: true`.
|
|
|
167
155
|
### Phase 7: Issue cleanup
|
|
168
156
|
Close issues that shipped fixes — only those. Skipped issues stay open.
|
|
169
157
|
|
|
158
|
+
Each issue carries exactly ONE what-landed comment — the fix summary Phase 3 posted, with the version added (`Shipped in v<version>: …`) if it lacks one. Then close without an additional comment:
|
|
159
|
+
|
|
170
160
|
```bash
|
|
171
161
|
for n in <fixed-issue-numbers-from-phase-3>; do
|
|
172
|
-
gh issue close "$n" -R "<owner>/<repo>" --reason completed
|
|
162
|
+
gh issue close "$n" -R "<owner>/<repo>" --reason completed
|
|
173
163
|
done
|
|
174
164
|
```
|
|
175
165
|
|
|
@@ -205,4 +195,4 @@ Collect specific issue numbers from Phase 3 sub-agent summaries — do not close
|
|
|
205
195
|
- [ ] Phase 6 (if releasing): version bumped, fix commits + release commit, annotated tag, scope matches private/public status
|
|
206
196
|
- [ ] Phase 7 (if releasing): fixed issues closed; skipped issues remain open
|
|
207
197
|
- [ ] Post-workflow verification: `git ls-remote --tags origin`, `npm view <pkg>@<version>` if public, GH release artifacts attached
|
|
208
|
-
- [ ] Tag/release quality review:
|
|
198
|
+
- [ ] Tag/release quality review: `bun run release:github -- --check` passed before the push; no marketing adjectives, issue backlinks present
|
|
@@ -4,7 +4,7 @@ description: >
|
|
|
4
4
|
Workflow for landing known work (handoff document findings, tracked GH issues, observed gaps) and shipping it: fix → optional simplify and field-test verification → wrap-up → release across one or more MCP server projects. Generalizes "I have known issues to fix and ship" regardless of how the issues were surfaced. Chains the `field-test`, `report-issue-local`, `code-simplifier`, `git-wrapup`, and `release-and-publish` skills. Read `../SKILL.md` first for the universal rules and sub-agent strategy.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "1.
|
|
7
|
+
version: "1.2"
|
|
8
8
|
audience: external
|
|
9
9
|
type: workflow
|
|
10
10
|
---
|
|
@@ -20,7 +20,7 @@ The input varies but the workflow is the same. Read the inputs into a common sha
|
|
|
20
20
|
| Source | Shape it as |
|
|
21
21
|
|:---|:---|
|
|
22
22
|
| Handoff document (numbered findings, repro steps, acceptance criteria) | Validate each finding live in Phase 1a; file each valid one as a GH issue via `report-issue-local`; skip invalidated findings |
|
|
23
|
-
| GH issues already filed | Use as-is. Read each with `gh issue view N --comments`
|
|
23
|
+
| GH issues already filed | Use as-is. Read each with `gh issue view N` (the body) and `gh issue view N --comments` (the thread — the body alone misses clarifications and decision updates) |
|
|
24
24
|
| Observed gap or casual report ("I noticed this", "fix the description on tool X") | If material enough to ship in a release, file a GH issue first to capture rationale and create an audit trail. Trivial typo-fix-and-ship can skip the issue step. |
|
|
25
25
|
|
|
26
26
|
The validation/filing step is the difference between "input is a hypothesis" (handoff) and "input is verified" (tracked GH issues). The rest of the workflow is identical.
|
|
@@ -47,7 +47,7 @@ For unsourced QA — where the bugs are unknown until you test — use `field-te
|
|
|
47
47
|
|
|
48
48
|
Per target:
|
|
49
49
|
|
|
50
|
-
1. **Identify issues** — collect GH issue numbers to fix, the handoff document, or the explicit gap description. Read each issue with `gh issue view N --comments`
|
|
50
|
+
1. **Identify issues** — collect GH issue numbers to fix, the handoff document, or the explicit gap description. Read each issue with `gh issue view N` and `gh issue view N --comments` — body, then thread.
|
|
51
51
|
2. **Clean working tree** — `git status --short` must be empty
|
|
52
52
|
3. **Current version** — `git describe --tags --abbrev=0`, `grep '"version"' package.json`
|
|
53
53
|
4. **Repo visibility** — `gh repo view --json visibility -q '.visibility'`. Determines wrap-up scope.
|
|
@@ -62,7 +62,7 @@ Each phase's Objective column is the goal state per target — the verifiable en
|
|
|
62
62
|
| 1a | Validate (conditional) | Each handoff finding field-tested live; valid ones filed as GH issues; invalidated ones reported back with reason. If zero validate, workflow stops | one sub-agent per target | **barrier** — cross-target synthesis: orchestrator confirms validated findings before fix proceeds (or stops workflow if zero validate) |
|
|
63
63
|
| 1b | Fix | Per target: targeted issues fixed in source, tests updated/added, `devcheck` + `rebuild` + `test` green, each fixed issue commented with fix details, working tree dirty for review | parallel fanout (one sub-agent per target — hard constraint) | **barrier** — orchestrator reviews diffs before verify (explicit gate in checklist) |
|
|
64
64
|
| 2 | Verify | Per target: full diff cold-reviewed; simplified if warranted; each fix re-exercised against the running server with actual tool output in the summary | parallel fanout | **barrier** — orchestrator reviews simplified diff and verified outputs; release authorization required |
|
|
65
|
-
| 3 | Wrap-up + release | Per target: fixes
|
|
65
|
+
| 3 | Wrap-up + release | Per target: fixes grouped into one commit per concern (a file never splits across commits) with a release commit on top; annotated tag passing `bun run release:github -- --check` (flat bullets with issue backlinks, changelog link last); published per repo visibility | parallel fanout (Bash git only) | gate-free |
|
|
66
66
|
| 4 | Issue cleanup | Every shipped issue closed (reason: completed) carrying exactly one what-landed comment that cites the version | orchestrator (serial) | — |
|
|
67
67
|
|
|
68
68
|
Phase 1a is conditional — only runs when the input is a handoff document or otherwise unvalidated. When the input is already tracked GH issues, skip directly to Phase 1b. The release portion of Phase 3 is conditional on user authorization to ship.
|
|
@@ -89,7 +89,7 @@ If zero findings validate, report to the user and stop the workflow.
|
|
|
89
89
|
**One sub-agent per target — hard constraint** (no file-locking; concurrent edits to the same `src/` conflict).
|
|
90
90
|
|
|
91
91
|
Each sub-agent:
|
|
92
|
-
1. Reads all open issues for its target via `gh issue view N --comments` (
|
|
92
|
+
1. Reads all open issues for its target via `gh issue view N` and `gh issue view N --comments` (body, then thread — the body alone misses clarifications)
|
|
93
93
|
2. **Validates each issue against source code** — the issue's analysis or proposed approach may be wrong; sub-agent applies judgment about the right fix and notes any deviation in its GH comment
|
|
94
94
|
3. Prioritizes: security → crashes → bugs → enhancements → docs/chore
|
|
95
95
|
4. Implements fixes using the best modern approach (the GH issue is input, not a spec)
|
|
@@ -162,7 +162,7 @@ If no what-landed comment exists yet, the version belongs in that one comment ("
|
|
|
162
162
|
| 5 | Code-simplify removes intentional complexity | Orchestrator gate after Phase 2 reviews the full diff |
|
|
163
163
|
| 6 | Wrap-up sub-agent collapses multi-fix diff into one commit | Phase 3 prompt enumerates the commit structure |
|
|
164
164
|
| 7 | Wrap-up sub-agent makes unplanned intermediate commits outside the planned structure | Prompt defines exact commit shape; agents must not invent extras |
|
|
165
|
-
| 8 | Reading `gh issue view N` alone misses thread context where decisions were updated | Always
|
|
165
|
+
| 8 | Reading `gh issue view N` alone misses thread context where decisions were updated; `--comments` alone prints no body without a TTY | Always run both |
|
|
166
166
|
| 9 | MCP Registry returns 502 transiently during publish | Retry up to 2x with backoff |
|
|
167
167
|
| 10 | Phase 1a sub-agent validates an issue that's actually a misunderstanding | Sub-agent must field-test, not just read the claim — live verification catches false positives |
|
|
168
168
|
|
|
@@ -178,4 +178,4 @@ If no what-landed comment exists yet, the version belongs in that one comment ("
|
|
|
178
178
|
- [ ] Phase 3: published per scope (push, npm if public, MCP Registry if applicable, GH release, Docker if applicable)
|
|
179
179
|
- [ ] Phase 4: shipped issues closed, one what-landed comment each; skipped issues remain open
|
|
180
180
|
- [ ] Post-workflow verification: `git ls-remote --tags origin`, `npm view <pkg>@<version>` if public, GH release artifacts attached
|
|
181
|
-
- [ ] Tag/release quality review:
|
|
181
|
+
- [ ] Tag/release quality review: `bun run release:github -- --check` passed before the push; no marketing adjectives, issue backlinks present
|