@cyanheads/mcp-ts-core 0.13.5 → 0.13.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (157) hide show
  1. package/AGENTS.md +6 -6
  2. package/CLAUDE.md +6 -6
  3. package/README.md +55 -52
  4. package/biome.json +2 -2
  5. package/changelog/0.13.x/0.13.6.md +49 -0
  6. package/changelog/0.13.x/0.13.7.md +77 -0
  7. package/config/tsconfig.base.json +2 -2
  8. package/dist/config/index.d.ts.map +1 -1
  9. package/dist/config/index.js +42 -11
  10. package/dist/config/index.js.map +1 -1
  11. package/dist/core/app.d.ts.map +1 -1
  12. package/dist/core/app.js +21 -4
  13. package/dist/core/app.js.map +1 -1
  14. package/dist/core/context.d.ts +9 -1
  15. package/dist/core/context.d.ts.map +1 -1
  16. package/dist/core/context.js +4 -13
  17. package/dist/core/context.js.map +1 -1
  18. package/dist/core/worker.d.ts.map +1 -1
  19. package/dist/core/worker.js +7 -1
  20. package/dist/core/worker.js.map +1 -1
  21. package/dist/linter/rules/enrichment-rules.d.ts +5 -4
  22. package/dist/linter/rules/enrichment-rules.d.ts.map +1 -1
  23. package/dist/linter/rules/enrichment-rules.js +99 -22
  24. package/dist/linter/rules/enrichment-rules.js.map +1 -1
  25. package/dist/linter/rules/error-contract-rules.d.ts +46 -10
  26. package/dist/linter/rules/error-contract-rules.d.ts.map +1 -1
  27. package/dist/linter/rules/error-contract-rules.js +180 -27
  28. package/dist/linter/rules/error-contract-rules.js.map +1 -1
  29. package/dist/linter/rules/format-parity-rules.js +1 -1
  30. package/dist/linter/rules/format-parity-rules.js.map +1 -1
  31. package/dist/linter/rules/index.d.ts +1 -1
  32. package/dist/linter/rules/index.d.ts.map +1 -1
  33. package/dist/linter/rules/index.js +1 -1
  34. package/dist/linter/rules/index.js.map +1 -1
  35. package/dist/linter/rules/resource-rules.d.ts.map +1 -1
  36. package/dist/linter/rules/resource-rules.js +2 -1
  37. package/dist/linter/rules/resource-rules.js.map +1 -1
  38. package/dist/linter/rules/tool-rules.d.ts.map +1 -1
  39. package/dist/linter/rules/tool-rules.js +2 -1
  40. package/dist/linter/rules/tool-rules.js.map +1 -1
  41. package/dist/mcp-server/handlerContext.d.ts +6 -0
  42. package/dist/mcp-server/handlerContext.d.ts.map +1 -1
  43. package/dist/mcp-server/handlerContext.js +3 -0
  44. package/dist/mcp-server/handlerContext.js.map +1 -1
  45. package/dist/mcp-server/prompts/prompt-registration.d.ts.map +1 -1
  46. package/dist/mcp-server/prompts/prompt-registration.js +6 -3
  47. package/dist/mcp-server/prompts/prompt-registration.js.map +1 -1
  48. package/dist/mcp-server/tools/utils/toolHandlerFactory.d.ts.map +1 -1
  49. package/dist/mcp-server/tools/utils/toolHandlerFactory.js +5 -1
  50. package/dist/mcp-server/tools/utils/toolHandlerFactory.js.map +1 -1
  51. package/dist/mcp-server/transports/http/httpErrorHandler.d.ts.map +1 -1
  52. package/dist/mcp-server/transports/http/httpErrorHandler.js +15 -5
  53. package/dist/mcp-server/transports/http/httpErrorHandler.js.map +1 -1
  54. package/dist/storage/core/IStorageProvider.d.ts +5 -2
  55. package/dist/storage/core/IStorageProvider.d.ts.map +1 -1
  56. package/dist/storage/core/providerHelpers.d.ts +29 -8
  57. package/dist/storage/core/providerHelpers.d.ts.map +1 -1
  58. package/dist/storage/core/providerHelpers.js +49 -11
  59. package/dist/storage/core/providerHelpers.js.map +1 -1
  60. package/dist/storage/providers/cloudflare/d1Provider.js +4 -4
  61. package/dist/storage/providers/cloudflare/d1Provider.js.map +1 -1
  62. package/dist/storage/providers/cloudflare/kvProvider.d.ts +2 -0
  63. package/dist/storage/providers/cloudflare/kvProvider.d.ts.map +1 -1
  64. package/dist/storage/providers/cloudflare/kvProvider.js +11 -9
  65. package/dist/storage/providers/cloudflare/kvProvider.js.map +1 -1
  66. package/dist/storage/providers/cloudflare/r2Provider.d.ts.map +1 -1
  67. package/dist/storage/providers/cloudflare/r2Provider.js +8 -5
  68. package/dist/storage/providers/cloudflare/r2Provider.js.map +1 -1
  69. package/dist/storage/providers/fileSystem/fileSystemProvider.d.ts +1 -0
  70. package/dist/storage/providers/fileSystem/fileSystemProvider.d.ts.map +1 -1
  71. package/dist/storage/providers/fileSystem/fileSystemProvider.js +10 -8
  72. package/dist/storage/providers/fileSystem/fileSystemProvider.js.map +1 -1
  73. package/dist/storage/providers/inMemory/inMemoryProvider.d.ts +5 -0
  74. package/dist/storage/providers/inMemory/inMemoryProvider.d.ts.map +1 -1
  75. package/dist/storage/providers/inMemory/inMemoryProvider.js +9 -5
  76. package/dist/storage/providers/inMemory/inMemoryProvider.js.map +1 -1
  77. package/dist/storage/providers/supabase/supabaseProvider.d.ts.map +1 -1
  78. package/dist/storage/providers/supabase/supabaseProvider.js +5 -1
  79. package/dist/storage/providers/supabase/supabaseProvider.js.map +1 -1
  80. package/dist/testing/index.d.ts +6 -4
  81. package/dist/testing/index.d.ts.map +1 -1
  82. package/dist/testing/index.js +6 -4
  83. package/dist/testing/index.js.map +1 -1
  84. package/dist/types-global/errors.d.ts +18 -0
  85. package/dist/types-global/errors.d.ts.map +1 -1
  86. package/dist/utils/formatting/partialResult.d.ts +28 -2
  87. package/dist/utils/formatting/partialResult.d.ts.map +1 -1
  88. package/dist/utils/formatting/partialResult.js +46 -2
  89. package/dist/utils/formatting/partialResult.js.map +1 -1
  90. package/dist/utils/internal/error-handler/errorHandler.d.ts +8 -2
  91. package/dist/utils/internal/error-handler/errorHandler.d.ts.map +1 -1
  92. package/dist/utils/internal/error-handler/errorHandler.js +27 -16
  93. package/dist/utils/internal/error-handler/errorHandler.js.map +1 -1
  94. package/dist/utils/internal/error-handler/mappings.d.ts +1 -0
  95. package/dist/utils/internal/error-handler/mappings.d.ts.map +1 -1
  96. package/dist/utils/internal/error-handler/mappings.js +1 -0
  97. package/dist/utils/internal/error-handler/mappings.js.map +1 -1
  98. package/dist/utils/internal/performance.d.ts +5 -1
  99. package/dist/utils/internal/performance.d.ts.map +1 -1
  100. package/dist/utils/internal/performance.js +13 -8
  101. package/dist/utils/internal/performance.js.map +1 -1
  102. package/dist/utils/security/sanitization.d.ts +15 -15
  103. package/dist/utils/security/sanitization.d.ts.map +1 -1
  104. package/dist/utils/security/sanitization.js +108 -88
  105. package/dist/utils/security/sanitization.js.map +1 -1
  106. package/dist/utils/telemetry/instrumentation.d.ts +6 -2
  107. package/dist/utils/telemetry/instrumentation.d.ts.map +1 -1
  108. package/dist/utils/telemetry/instrumentation.js +23 -8
  109. package/dist/utils/telemetry/instrumentation.js.map +1 -1
  110. package/framework-skills/add-app-tool/SKILL.md +12 -18
  111. package/framework-skills/add-prompt/SKILL.md +3 -1
  112. package/framework-skills/add-provider/SKILL.md +14 -4
  113. package/framework-skills/add-resource/SKILL.md +3 -3
  114. package/framework-skills/add-service/SKILL.md +5 -2
  115. package/framework-skills/add-tool/SKILL.md +25 -8
  116. package/framework-skills/api-canvas/SKILL.md +2 -2
  117. package/framework-skills/api-config/SKILL.md +4 -3
  118. package/framework-skills/api-context/SKILL.md +8 -5
  119. package/framework-skills/api-errors/SKILL.md +25 -5
  120. package/framework-skills/api-linter/SKILL.md +95 -22
  121. package/framework-skills/api-telemetry/SKILL.md +9 -4
  122. package/framework-skills/api-testing/SKILL.md +21 -13
  123. package/framework-skills/api-utils/SKILL.md +3 -3
  124. package/framework-skills/api-utils/references/security.md +7 -6
  125. package/framework-skills/code-simplifier/SKILL.md +31 -18
  126. package/framework-skills/design-mcp-server/SKILL.md +62 -35
  127. package/framework-skills/git-wrapup/SKILL.md +16 -10
  128. package/framework-skills/maintenance/SKILL.md +2 -2
  129. package/framework-skills/orchestrations/SKILL.md +1 -1
  130. package/framework-skills/orchestrations/workflows/greenfield-build.md +15 -8
  131. package/framework-skills/polish-docs-meta/SKILL.md +2 -2
  132. package/framework-skills/polish-docs-meta/references/package-meta.md +1 -1
  133. package/framework-skills/polish-docs-meta/references/readme.md +3 -3
  134. package/framework-skills/release-and-publish/SKILL.md +6 -4
  135. package/framework-skills/release-pr-review/SKILL.md +18 -1
  136. package/framework-skills/report-issue-framework/SKILL.md +2 -2
  137. package/framework-skills/report-issue-local/SKILL.md +3 -3
  138. package/framework-skills/security-pass/SKILL.md +11 -3
  139. package/framework-skills/tool-defs-analysis/SKILL.md +3 -3
  140. package/package.json +16 -37
  141. package/scripts/lint-mcp.ts +43 -4
  142. package/templates/.env.example +3 -1
  143. package/templates/.github/ISSUE_TEMPLATE/feature_request.yml +1 -1
  144. package/templates/AGENTS.md +3 -3
  145. package/templates/CLAUDE.md +3 -3
  146. package/templates/Dockerfile +4 -4
  147. package/templates/devcheck.config.json +1 -0
  148. package/templates/package.json +4 -4
  149. package/templates/src/mcp-server/prompts/definitions/echo.prompt.ts +2 -4
  150. package/templates/src/mcp-server/resources/definitions/echo-app-ui.app-resource.ts +51 -14
  151. package/templates/src/mcp-server/resources/definitions/echo.resource.ts +1 -1
  152. package/templates/src/mcp-server/tools/definitions/echo-app.app-tool.ts +2 -3
  153. package/templates/src/mcp-server/tools/definitions/echo.tool.ts +8 -2
  154. package/dist/utils/telemetry/index.d.ts +0 -12
  155. package/dist/utils/telemetry/index.d.ts.map +0 -1
  156. package/dist/utils/telemetry/index.js +0 -12
  157. package/dist/utils/telemetry/index.js.map +0 -1
@@ -4,7 +4,7 @@ description: >
4
4
  Design the tool surface, resources, and service layer for a new MCP server. Use when starting a new server, planning a major feature expansion, or when the user describes a domain/API they want to expose via MCP. Produces a design doc at docs/design.md that drives implementation.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "2.28"
7
+ version: "2.29"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -26,6 +26,7 @@ Gather before designing. Ask the user if not obvious from context:
26
26
  2. **Data sources / source of truth** — APIs, databases, file systems, external services? Or is the server itself the source (in-memory state, pure computation, local-only utility, embedded model)?
27
27
  3. **Target users** — what will the LLM (and its human) be trying to accomplish?
28
28
  4. **Scope constraints** — read-only? write access? admin operations? what's off-limits?
29
+ 5. **Deployment** — local stdio, hosted HTTP, Cloudflare Workers? The answer gates primitives: DataCanvas and `MirrorService` don't run on Workers, and a tool that asks the caller for input mid-call needs a stateful HTTP session (see the Client round-trip row in Step 3).
29
30
 
30
31
  If the domain has a public API, read its docs before designing. For internal-only servers, skip API research and go straight to user goals. Don't design from vibes either way.
31
32
 
@@ -63,6 +64,7 @@ Research inline by default — fetch docs, read SDK readmes, confirm assumptions
63
64
  - Fetch API docs, confirm endpoint availability, auth methods, rate limits
64
65
  - Check for official SDKs or client libraries (npm packages)
65
66
  - Note any API quirks, pagination patterns, or data format considerations
67
+ - Read the terms of use: whether storing, caching, or redistributing the data is permitted, whether AI or LLM use is, and what attribution the data carries. Note the credential model too — keyless, one operator key, or a key each user supplies. The terms decide whether the server can be hosted for others at all, whether a local mirror is allowed, and what the server instructions must credit.
66
68
 
67
69
  When research is genuinely parallelizable (multiple independent APIs, several SDKs to evaluate), spawn background agents for the independent legs while you proceed with domain mapping. Skip the overhead for a single API — just read it yourself.
68
70
 
@@ -75,11 +77,14 @@ When research is genuinely parallelizable (multiple independent APIs, several SD
75
77
  - **Error shapes** — trigger real 400/404/429 responses to see the actual error format, not just what docs claim.
76
78
  - **Unknown-param behavior** — send one deliberately misspelled parameter. If the API silently ignores it (plausible but unfiltered results instead of an error), every typo'd or unverified param name becomes a silent-wrongness bug — the service layer then needs a strict allowlist of confirmed spellings, and new filters require a probe before they ship.
77
79
  - **Omission semantics** — for each major optional parameter, check what omitting it actually returns. Some APIs default to the intuitive scope; others silently widen (all historical versions, all statuses, global instead of regional). A default that changes result *meaning* becomes a server-side default plus an echoed output field, not something left to the agent.
80
+ - **Branch frequency, for bundled or bulk data** — when the "API" is a dataset the server ships, probe not only whether each record shape exists but what *fraction* of records takes it: count the records where an optional field is absent, where a value is an array instead of a scalar, where an id maps to many keys rather than one, where a status is missing. A shape found in 0.1% of records is an edge case to handle; one found in 70% is the main path, and a design that specifies only the scalar case leaves the implementer to invent the common one.
78
81
 
79
82
  **Stopping condition:** at minimum, probe one list/search endpoint, one single-item GET, one error case (force a 404 or 400), and one unknown-param request. For large APIs with many resource types, add one probe per major noun. Stop when the response shapes and error envelope are confirmed.
80
83
 
81
84
  This step prevents building a service layer against assumed response shapes that don't match reality.
82
85
 
86
+ Probe with real data; write the design with made-up data. Every example value that reaches `docs/design.md` — names, emails, phone numbers, hosts, IP addresses, identifiers tied to a person — is synthetic, and no key, token, or private hostname appears in it. The doc is tracked, and git history keeps whatever its first commit carried.
87
+
83
88
  ### 2. Map User Goals, Then Domain Operations
84
89
 
85
90
  Start with **user goals**, not endpoints. Enumerate the outcomes an agent (and its human) will actually try to accomplish with this server — usually 3–10, scaled to domain size. These drive the workflow tools that form the spine of the surface. Endpoint-inventory-first design produces 1:1 API mirrors; goal-first design produces tools agents reach for. For internal-only servers, goals map to capabilities rather than endpoints — e.g., "format markdown to GFM," "tokenize text by model," "compute file hash."
@@ -111,7 +116,7 @@ The user-goal list shapes the tool surface; the operation list fills in the gaps
111
116
  | **App Tool** | **Rare — default to a standard tool.** Only when a human will actively interact with the result in real time *and* the target client supports MCP Apps. Most clients are tool-only and most agent workflows are read-by-LLM, not viewed-by-human. App tools add an iframe + CSP, `app.ontoolresult`/`callServerTool` plumbing, host-context wiring, and a `format()` text twin that still has to be content-complete (since most clients only see that). Two surfaces to keep in sync, two failure modes per change. | Dense tabular state a human scrubs through; form-based human approval in an MCP Apps-capable client |
112
117
  | **Resource** | *Additionally* expose as a resource when the data is addressable by stable URI, read-only, and useful as injectable context. | Config, schemas, status, entity-by-ID lookups |
113
118
  | **Prompt** | Reusable message template that structures how the LLM approaches a task | Analysis framework, report template, review checklist |
114
- | **Client round-trip** | Not a registered primitive — a handler *returns* `ctx.requestInput(...)` to ask the client for what only it has, and is re-entered with the answer: a confirmation or form (`inputRequired.elicit`), an authorization or hosted-form URL (`inputRequired.elicitUrl`), the client model's judgment (`inputRequired.createMessage` — borrow the caller's model rather than bundling one), filesystem roots (`inputRequired.listRoots`). Design it into the tool that needs it; see Workflow tool safety and `api-context`. | Destructive-arm confirmation, OAuth consent, summarize-with-the-client's-model |
119
+ | **Client round-trip** | Not a registered primitive — a handler *returns* `ctx.requestInput(...)` to ask the client for what only it has, and is re-entered with the answer: a confirmation or form (`inputRequired.elicit`), an authorization or hosted-form URL (`inputRequired.elicitUrl`), the client model's judgment (`inputRequired.createMessage` — borrow the caller's model rather than bundling one), filesystem roots (`inputRequired.listRoots`). Design it into the tool that needs it; see Workflow tool safety and `api-context`. A server with any such tool declares `createApp({ sessionMode: { default: 'stateful', require: 'stateful' } })` — under stateless HTTP a 2025-era client's round trip is refused, so the tool is unusable rather than guarded. | Destructive-arm confirmation, OAuth consent, summarize-with-the-client's-model |
115
120
  | **Neither** | Internal detail, admin-only, not useful to an LLM | Token refresh, webhook setup, migrations |
116
121
 
117
122
  What the tool surface needs to cover depends on the server: a read-only research server has different economics than a CRUD project management server. Consider the domain, the expected agent workflows, whether it wraps one API or many, and what data relationships exist.
@@ -132,7 +137,7 @@ This is the highest-leverage step. Tool definitions — names, descriptions, par
132
137
 
133
138
  #### Tool shapes you'll encounter
134
139
 
135
- Most tools follow the `{server}_{verb}_{noun}` default — one focused responsibility, one clear verb, often (but not always) one upstream call. API-wrapping examples: `pubmed_search_articles`, `pubmed_fetch_articles`. Internal-only examples: `markdown_format_text`, `regex_test_pattern`, `tokens_count_text` — same naming convention, no external dep. Two variants warrant explicit design pressures of their own:
140
+ Most tools follow the `{server}_{verb}_{noun}` default — one focused responsibility, one clear verb, often (but not always) one upstream call. API-wrapping examples: `pubmed_search_articles`, `pubmed_fetch_articles`. Internal-only examples: `markdown_format_text`, `regex_test_pattern`, `tokens_count_text` — same naming convention, no external dep. Three variants warrant explicit design pressures of their own:
136
141
 
137
142
  | Shape | Purpose | Typical form | Examples |
138
143
  |:------|:--------|:-------------|:---------|
@@ -168,24 +173,29 @@ Two patterns:
168
173
 
169
174
  **Source fallback chains** — try sources in priority order, fall through on failure or empty results. Best when sources cover the *same corpus* with different depth or availability. The output should indicate which source provided the data so the agent (and human) can assess provenance. When the fallback changes what is being searched — a different corpus, different identifiers, different licensing — don't chain: expose the second source as a sibling tool so the agent chooses the corpus knowingly (the shipped `pubmed-mcp-server` keeps `pubmed_europepmc_search` separate for exactly this reason).
170
175
 
171
- **Multi-source fan-out** — query multiple sources in parallel, merge results. Best when sources provide complementary data about the same entity. Use `Promise.allSettled` so one failing source doesn't tank the whole call.
176
+ **Multi-source fan-out** — query multiple sources in parallel, merge results. Best when sources provide complementary data about the same entity. Use `Promise.allSettled` so one *unavailable* source doesn't tank the whole call — but a rejected input is wrong for every source. Rethrow an input-class rejection (`ValidationError`, `InvalidParams`) from any source instead of folding it into a per-source failure: a source that ignores the bad value would otherwise answer for it, and the agent reads a caller mistake as an outage to retry.
172
177
 
173
178
  ```ts
174
179
  // Handler pseudocode — indicator enrichment across threat intel sources
175
180
  async handler(input, ctx) {
176
- const [vt, abuse, greynoise] = await Promise.allSettled([
181
+ const settled = await Promise.allSettled([
177
182
  vtService.lookup(input.indicator),
178
183
  abuseIpService.check(input.indicator),
179
184
  greynoiseService.query(input.indicator),
180
185
  ]);
186
+ for (const r of settled) {
187
+ if (r.status === 'rejected' && isInputError(r.reason)) throw r.reason; // the caller's mistake, not an outage
188
+ }
189
+ const [vt, abuse, greynoise] = settled;
181
190
  return {
182
191
  indicator: input.indicator,
183
192
  sources: {
184
- virustotal: vt.status === 'fulfilled' ? vt.value : { error: vt.reason.message },
185
- abuseipdb: abuse.status === 'fulfilled' ? abuse.value : { error: abuse.reason.message },
186
- greynoise: greynoise.status === 'fulfilled' ? greynoise.value : { error: greynoise.reason.message },
193
+ // { status: 'ok', data } | { status: 'unavailable', error } — provenance per source
194
+ virustotal: toSourceResult(vt),
195
+ abuseipdb: toSourceResult(abuse),
196
+ greynoise: toSourceResult(greynoise),
187
197
  },
188
- // Server synthesizes a verdict from available data — the agent gets a conclusion, not raw API dumps
198
+ // Server synthesizes a verdict from the sources that answered — the agent gets a conclusion, not raw API dumps
189
199
  assessment: synthesizeVerdict(vt, abuse, greynoise),
190
200
  };
191
201
  }
@@ -222,12 +232,12 @@ const wrapupInstructions = tool('git_wrapup_instructions', {
222
232
  output: z.object({
223
233
  guidance: z.string()
224
234
  .describe('Markdown playbook content, tailored to current account state.'),
225
- diagnostics: z.record(z.unknown())
235
+ diagnostics: z.record(z.string(), z.unknown())
226
236
  .describe('Live state used to tailor the guidance (e.g., staged file count, branch divergence, recent commit cadence).'),
227
237
  nextToolSuggestions: z.array(z.object({
228
238
  toolName: z.string().describe('Tool to call next.'),
229
239
  reason: z.string().describe('Why this step is recommended given current state.'),
230
- args: z.record(z.unknown()).describe('Arguments pre-filled from diagnostics.'),
240
+ args: z.record(z.string(), z.unknown()).describe('Arguments pre-filled from diagnostics; {} when the tool takes none.'),
231
241
  })).describe('Recommended follow-up calls with arguments already populated.'),
232
242
  }),
233
243
  });
@@ -235,6 +245,10 @@ const wrapupInstructions = tool('git_wrapup_instructions', {
235
245
 
236
246
  Prior art: [`git_wrapup_instructions`](https://github.com/cyanheads/git-mcp-server) walks through staging, commit, and push with repo state inspected. If a server has recurring "how do I do X well given my state" questions, an instruction tool typically beats N topic-specific tools and duplicating guidance in tool descriptions.
237
247
 
248
+ **One suggestion shape, every server.** Each entry is exactly `{ toolName, reason, args }` — no per-server renames (`tool`, `rationale`, `suggestedArgs`, `arguments`), which force every client to special-case every server. `args` is always present (`{}` for a tool that takes none), and the array is always present, empty when nothing is worth suggesting. A tool on another server is never an entry: this server cannot know it is installed, so name it in the `guidance` prose instead.
249
+
250
+ **Data tools qualify on the same terms.** A data tool carries `nextToolSuggestions` when the right next call depends on its own result and the arguments come from that result — a status sweep that finds a degraded vendor, a connect call that learns which services the upstream supports. A fixed chain (search → get by the returned ID) does not: the IDs in the result and the server instructions already carry it, and a suggestion on every call is token noise.
251
+
238
252
  **Suggestions are scoped to what this deployment registers.** A `nextToolSuggestions` entry is an executable call, so it is only correct when its target is enabled under the same configuration — a tool wrapped in `disabledTool()` is absent from `tools/list`, and a suggestion naming it hands the agent a call that fails on dispatch. Build the array from the same config the registration reads, and when the target is off, drop the entry rather than the explanation: `guidance` can still say the capability is unavailable in this deployment and what to do instead. The audit and a worked example live under *Feature-flagged tools* in `add-tool/SKILL.md`.
239
253
 
240
254
  #### Reference tools
@@ -287,7 +301,8 @@ Descriptions should be as long as needed — concise but complete. Don't artific
287
301
  Every `.describe()` is prompt text the LLM reads. Parameters should convey: what the value is, what it affects, and (where non-obvious) how to use it well.
288
302
 
289
303
  - **Constrain the type.** Enums and literals over free strings. Regex validation for formatted IDs. Ranges for numeric bounds.
290
- - **The input root is already strict.** `tool()` applies `.strict()` at the root and advertises `additionalProperties: false`, so an unknown top-level key is rejected by name instead of silently stripped; nested objects still strip unless made strict themselves. Declare `.passthrough()` only on a tool that deliberately proxies arbitrary upstream parameters (a raw-query tool), and say so in its description.
304
+ - **The input root is already strict.** `tool()` applies `.strict()` at the root and advertises `additionalProperties: false`, so an unknown top-level key is rejected by name instead of silently stripped; nested objects still strip unless made strict themselves. Open the root with `.passthrough()` (Zod 4 also spells it `.loose()`; `tool()` honors either) only on a tool that deliberately proxies arbitrary upstream parameters (a raw-query tool), and say so in its description.
305
+ - **A blank optional string is unset.** Form-based clients send every optional field they display as `""`. Treat the blank as omitted — left off the upstream request, never forwarded as `param=`, which some APIs read differently from omission — and never design a `.min(1)` onto an optional field to catch it. `add-tool` has the schema pattern that keeps a validator on the field.
291
306
  - **Use JSON-Schema-serializable types only.** The MCP SDK serializes schemas to JSON Schema for `tools/list`. Types like `z.custom()`, `z.date()`, `z.transform()`, `z.bigint()`, `z.symbol()`, `z.void()`, `z.map()`, `z.set()` throw at runtime. Use structural equivalents (e.g., `z.string().describe('ISO 8601 date')` instead of `z.date()`).
292
307
  - **Explain costs and tradeoffs** when a parameter choice has meaningful consequences.
293
308
  - **Name alternative approaches** when a simpler path exists.
@@ -331,6 +346,7 @@ The output schema and `format` function control what the LLM reads back. Design
331
346
  - **Include IDs and references for chaining.** If the agent might act on a result, return the identifiers it needs for follow-up tool calls.
332
347
  - **Curate vs. pass-through depends on domain.** Medical/scientific data — don't trim fields that could alter correctness. CRUD responses — return what the agent needs, not the full API payload. Match fidelity to consequence.
333
348
  - **Absent upstream data stays absent.** Sparse APIs omit fields; declare those output fields `.optional()` and render "Not available" in `format()` rather than coercing to `false`, `0`, or `""` — a fabricated value is worse than a gap.
349
+ - **Third-party text is data, and `format()` marks it.** When a field carries text other people wrote — posts, reviews, bios, comments, summaries, a user-named place or project — list those fields in the design and plan how `format()` sets them apart: a blockquote or fence for free text, and CR/LF flattened to a space wherever a value is interpolated inline (a heading, a bolded name, a `matched on "…"` line), so a newline in the value can't forge structure in `content[]`. `structuredContent` keeps the value verbatim. Say in the server instructions that this content is data, never instructions; `security-pass` audits the result.
334
350
  - **Image and audio bytes ride `ctx.content`, never `output`.** `ctx.content.image(data, mimeType)` / `.audio(...)` emit a `content[]` block once; `output` keeps the metadata the agent reasons over (dimensions, duration, a reference). Base64 in a typed output field ships the bytes twice.
335
351
  - **Surface what was done, not just results.** After a write operation, include the post-state so the LLM can chain without an extra round trip.
336
352
  - **When the effect lands after the call, wait for it by default.** Some actions are fire-and-forget at the wire — send a wake packet, trigger a job, dispatch a notification, provision a resource — and their immediate result ("sent", "queued", "accepted") answers nothing the agent asked; the agent wants to know whether the machine is up, the job ran, the resource exists. Design the tool to confirm: pre-probe the observable state (cheap; if it is already in the target state, skip the action and say so), act, then poll with early return until the state is observed or a bounded window elapses. Express the window as **one numeric parameter with a default** (`wait_for_s: 30`), where `0` means act and return — not a boolean plus a timeout, which is two parameters for one decision and an awkward default. Every terminal outcome is a *result*, never a throw: `already_<state>`, `<state>` (with elapsed time), `not_<state>` within the window (with `guidance` naming the read-only check tool to re-poll), and `unverified` (window `0`, or nothing to probe). Size the default to cover the common cases while staying inside client tool timeouts, and pair the action with a read-only sibling that probes the same state so an agent can re-check without re-acting.
@@ -341,7 +357,7 @@ The output schema and `format` function control what the LLM reads back. Design
341
357
  - **Continuation is a designed field.** Truncation says the cap was hit; continuation says how to get the rest. Return an opaque `cursor` plus `has_more` (via `extractCursor`/`paginateArray` for local sets), and never invent page numbers over a cursor-based upstream — a page the agent can't ask for is a page it will never see.
342
358
  - **Spill big *analytical* results to a queryable surface.** When a tool's row set is something an agent would run SQL over *and* can exceed any reasonable context budget — paginated APIs, streamed exports, big query results — pair an inline preview with a `DataCanvas` table holding the full set (`spillover()` in `api-canvas`), and compute distributions or refinement hints across the full result, not the preview, so aggregate signal stays honest. The gates on when a canvas earns its keep are in Step 7.
343
359
  - **Outline one large *document* into sections.** When a single tool call returns one document-shaped record (not many rows) that can exceed context — a ~130KB FDA drug label, a big API entity dominated by a few fat fields — return a section *outline* (top-level keys + per-section byte size) instead of truncating, and let the agent re-call with `sections: [...]` to pull only what it needs. `outlineOnOverflow()` (`@cyanheads/mcp-ts-core/utils`) returns a `full | outline` result; pure measure + key-slice, so Cloudflare Workers-portable, unlike canvas-bound `spillover()`. Distinct from spillover on *shape*: spillover splits a row collection, this outlines one fat record. Schema shape and `format()` parity are in the `techniques` skill's `outline-on-overflow` reference.
344
- - **Mirror a bulk upstream instead of paginating it live.** When the server wraps a large or slow API whose corpus is queried far more than it changes, sync it once into a persistent local index and query that as the primary data path — not the live API per request. Match the backend to corpus size: below ~10⁴ rows → an in-memory index (server-level, no primitive); ~10⁴–10⁷ → the `MirrorService` (embedded SQLite + FTS5; declare a schema + a `sync` ingester via `defineMirror`/`sqliteMirrorStore`, then `runSync`/`query`, see `api-mirror`); above ~10⁷ → an external store. Distinct lifecycle from DataCanvas: a mirror is long-lived and cross-session, refreshed on a schedule; canvas is ephemeral and per-session.
360
+ - **Mirror a bulk upstream instead of paginating it live.** When the server wraps a large or slow API whose corpus is queried far more than it changes, sync it once into a persistent local index and query that as the primary data path — not the live API per request. Two gates come first. The upstream's terms must permit storing the data (Step 1), and a local index must reproduce what the upstream's *search* returns: an API that ranks server-side — field weighting, synonym or ontology expansion — can't be mirrored for search, because the local index answers the same query with different rows and nothing flags it. Exact-key lookups and structured filters mirror faithfully; for rate-limit relief on a ranked-search API, cache responses per request with a short TTL instead. Match the backend to corpus size: below ~10⁴ rows → an in-memory index (server-level, no primitive); ~10⁴–10⁷ → the `MirrorService` (embedded SQLite + FTS5; declare a schema + a `sync` ingester via `defineMirror`/`sqliteMirrorStore`, then `runSync`/`query`, see `api-mirror`); above ~10⁷ → an external store. Distinct lifecycle from DataCanvas: a mirror is long-lived and cross-session, refreshed on a schedule; canvas is ephemeral and per-session.
345
361
  - **Two client surfaces, both content-complete.** Different MCP clients forward different surfaces to the model: some (e.g., Claude Code) read `structuredContent` from `output`, others (e.g., Claude Desktop) read `content[]` from `format()`. `format()` is the markdown twin of `structuredContent`, not a summary — a thin `format()` that returns only a count or title leaves `content[]`-only clients blind (the `format-parity` lint catches this). Agent-facing context that is *not* domain payload — empty-result notices, the query as the server parsed it, echoed defaults, totals — goes in the `enrichment` block via `ctx.enrich(...)`, which reaches both surfaces automatically; hand-authored into `format()` text alone it reaches only one. Field-by-field rendering of that block is in the Design table's Enrichment row.
346
362
 
347
363
  #### Batch input design
@@ -382,7 +398,7 @@ When a tool wraps a complex query language or filter system, provide a simple sh
382
398
  // text_search handles the common case; query handles everything else
383
399
  text_search: z.string().optional()
384
400
  .describe('Convenience shortcut: full-text search across title and abstract. For structured filters or field-specific matching, use the query parameter instead.'),
385
- query: z.record(z.unknown()).optional()
401
+ query: z.record(z.string(), z.unknown()).optional()
386
402
  .describe('Full query object for structured filters. Supports operators: _eq, _gt, _and, _or, ...'),
387
403
  ```
388
404
 
@@ -407,18 +423,20 @@ Two params, two behaviors — keep them named distinctly:
407
423
 
408
424
  Errors are part of the tool's interface — design them during the design phase, not as an afterthought. Three aspects: **the contract** (which failures are public), **classification** (what error code), and **messaging** (what the LLM reads).
409
425
 
410
- **Declare a typed contract for domain failures.** When a tool has known failure modes the agent should plan around (`no_match`, `queue_full`, `vendor_down`), enumerate them as `errors: [{ reason, code, when, recovery, retryable? }]` on the definition. `recovery` is required metadata — the agent's next move when this failure fires (≥ 5 words, lint-validated; spread `ctx.recoveryFor('reason')` into the throw-site `data` to send it on the wire as `data.recovery.hint`). The framework types `ctx.fail(reason, …)` against the declared reason union (typos become TS errors) and auto-populates `data.reason` on the thrown error for stable observability. The error reaches clients with parity across both surfaces — `structuredContent.error` (Claude Code) and `content[]` text (Claude Desktop). Baseline codes (`InternalError`, `ServiceUnavailable`, `Timeout`, `ValidationError`, `SerializationError`) bubble from anywhere and don't need to be enumerated. See `api-errors` skill for the full pattern.
426
+ **Declare a typed contract for domain failures.** When a tool has known failure modes the agent should plan around (`no_match`, `queue_full`, `vendor_down`), enumerate them as `errors: [{ reason, code, when, recovery, retryable? }]` on the definition. `recovery` is required metadata — the agent's next move when this failure fires (≥ 5 words, lint-validated; spread `ctx.recoveryFor('reason')` into the throw-site `data` to send it on the wire as `data.recovery.hint`). The framework types `ctx.fail(reason, …)` against the declared reason union (typos become TS errors) and auto-populates `data.reason` on the thrown error for stable observability. The error reaches clients with parity across both surfaces — `structuredContent.error` (Claude Code) and `content[]` text (Claude Desktop). Baseline codes (`InternalError`, `ServiceUnavailable`, `Timeout`, `ValidationError`, `SerializationError`, `RequestCancelled`) bubble from anywhere and don't need to be enumerated. Mark an entry the service layer throws, rather than the handler, with `thrownBy: 'service'` so the conformance lint doesn't report it as a reason the handler never raises. See `api-errors` skill for the full pattern.
411
427
 
412
428
  **Classify errors by origin.** Different error sources need different codes and different recovery guidance. Map the failure modes for each tool during design:
413
429
 
414
430
  | Origin | Examples | Error code | Agent can recover? |
415
431
  |:-------|:---------|:-----------|:-------------------|
416
432
  | **Client input** | Bad ID format, invalid params, missing required field, out-of-range value | `ValidationError` | Yes — fix the input and retry |
417
- | **Upstream API** | 5xx, timeout, network error | `ServiceUnavailable` | Maybe — retry later, or the upstream is down |
433
+ | **Upstream API** | 5xx, network error | `ServiceUnavailable` | Maybe — retry later, or the upstream is down |
434
+ | **Timeout** | Upstream 408/504, a fetch that ran out its timeout, an exhausted retry deadline | `Timeout` | Maybe — retry, or narrow the request so it finishes sooner |
418
435
  | **Rate limit** | 429, quota exhausted, queue full | `RateLimited` (`retryable: true`; `withRetry` honors `Retry-After`) | Yes — wait, then retry or reduce frequency |
419
436
  | **Not found** | Valid ID format but entity doesn't exist | `NotFound` (or `ValidationError` if ambiguous) | Yes — check the ID, try a search |
420
437
  | **Conflict** | Duplicate key, version mismatch, concurrent modification on a write | `Conflict` | Yes — re-read current state, then retry with it |
421
- | **Auth/permissions** | Insufficient scopes, expired token | `Forbidden` / `Unauthorized` | Maybe — escalate or re-auth |
438
+ | **Caller auth** | Insufficient scopes, expired token, a rejected key the caller supplied | `Forbidden` / `Unauthorized` | Maybe — escalate or re-auth |
439
+ | **Server credential** | The server's own upstream key is missing, or the upstream rejects it (401/403) | `ConfigurationError` — translated in the service, since the automatic status mapping yields `Unauthorized`/`Forbidden`, which read as the caller's credentials | No — the operator fixes it; the `recovery` names the env var |
422
440
  | **Server internal** | Parse failure, missing config, unexpected state | `InternalError` | No — server-side issue |
423
441
 
424
442
  (`InvalidParams` also exists — the framework's `parseToolArguments` emits it when input fails Zod schema validation before the handler runs. Anything the handler itself throws about inputs uses `ValidationError`.)
@@ -465,15 +483,15 @@ Summarize each tool:
465
483
 
466
484
  | Aspect | Decision |
467
485
  |:-------|:---------|
468
- | **Name** | Lowercase snake_case with a canonical server prefix. **3 segments is the strong default** (`{server}_{verb}_{noun}` — e.g., `pubmed_search_articles`, `clinicaltrials_find_eligible`). **2 is fine when the operation name is canonical** and no noun adds signal (`git_pull`, `git_status` — "pull" already implies the remote). Don't invent a word to pad to 3. **4 is fine when the noun is inherently two words** (`openfda_search_device_clearances`) or the prefix is multi-part. The prefix is judged on clarity, not length: the brand name or the plain well-known word for the domain both pass (`pubmed_`, `patents_`, `earthquake_`); an abbreviation fails only when it reads as something else out of context (`loc_` → lines of code, `ct_` → CT scan). The verb+noun pair should be unambiguous within the server — if two tools could plausibly share a name, the noun isn't specific enough (`read_fulltext` not `read_text` when structured metadata is a separate concept). **Treat name length as a scope smell only when** the extra segment is the *verb* overreaching (e.g., `foo_create_and_send_notification` → split or use modes). |
486
+ | **Name** | Lowercase snake_case with a canonical server prefix. **3 segments is the strong default** (`{server}_{verb}_{noun}` — e.g., `pubmed_search_articles`, `clinicaltrials_find_eligible`). **2 is fine only when the verb is a complete action whose object the domain implies** (`git_pull`, `git_push`, `git_status`, `git_commit` — the remote, working tree, or repo is implicit); don't invent a word to pad those to 3. A verb that takes a caller-specified object — `search`, `find`, `get`, `list`, `query`, `fetch`, `connect`, `create`, `update` — always carries its noun (`ontology_search` → `ontology_search_terms`, `geofeatures_connect` → `geofeatures_connect_endpoint`). Smell test: if `{server}_{verb}` leaves "…what?" unanswered, the noun is missing. **4 is fine when the noun is inherently two words** (`openfda_search_device_clearances`) or the prefix is multi-part. The prefix is judged on clarity, not length: the brand name or the plain well-known word for the domain both pass (`pubmed_`, `patents_`, `earthquake_`); an abbreviation fails only when it reads as something else out of context (`loc_` → lines of code, `ct_` → CT scan). The verb+noun pair should be unambiguous within the server — if two tools could plausibly share a name, the noun isn't specific enough (`read_fulltext` not `read_text` when structured metadata is a separate concept). **Treat name length as a scope smell only when** the extra segment is the *verb* overreaching (e.g., `foo_create_and_send_notification` → split or use modes). |
469
487
  | **Granularity** | Scope each tool to one coherent agent action. The implementation can be a single API call (`pubmed_search_articles`), a multi-step workflow, or internal-only — match the unit to the work, don't constrain by call count. |
470
488
  | **Description** | Concrete capability statement. Add operational guidance (prerequisites, constraints, gotchas) when non-obvious. |
471
489
  | **Input schema** | `.describe()` on every field. Constrained types (enums, literals, regex). Explain costs/tradeoffs of parameter choices. |
472
490
  | **Output schema** | Designed for the LLM's next action. Include chaining IDs. Communicate filtering. Post-write state where useful. |
473
491
  | **Errors** | Declare domain failure modes as a typed contract (`errors: [{ reason, code, when, recovery, retryable? }]`) so `ctx.fail` is type-checked and capable clients can preview failures via `tools/list`. Every `recovery` string follows the no-dead-ends rule — it names the next tool call. |
474
- | **Enrichment** | The success-path counterpart to `errors`: declare the agent-facing context fields the handler populates via `ctx.enrich(...)` — zero-hit notice, echoed defaults, totals, truncation — with a kind-tag (`notice`/`total`/`echo`/`delta`) where one fits and an `enrichmentTrailer.render` for any structured field. Keys stay disjoint from `output`. |
492
+ | **Enrichment** | The success-path counterpart to `errors`: declare the agent-facing context fields the handler populates via `ctx.enrich(...)` — zero-hit notice, echoed defaults, totals, truncation — with a kind-tag (`notice`/`total`/`echo`/`delta`) where one fits and an `enrichmentTrailer.render` for any structured field. Keys stay disjoint from `output`. Declare a field required only when every path writes it — a required field one branch skips fails the output parse on every call that takes another branch. |
475
493
  | **Annotations** | `readOnlyHint`, `destructiveHint`, `idempotentHint`, `openWorldHint`. Helps clients auto-approve safely. `destructiveHint` defaults to **true** on any tool that isn't read-only, so a benign write must set `destructiveHint: false` explicitly; a read-only tool omits it entirely (`annotation-coherence` lint). |
476
- | **Auth scopes** | `tool:<snake_tool_name>:<verb>` or `resource:<kebab-resource-name>:<verb>` (e.g., `tool:inventory_search:read`, `resource:echo-app-ui:read`). Domain-led `<domain>:<verb>` (e.g., `inventory:read`) is an acceptable alternative — pick one convention per server and stay consistent. Skip when the server runs `MCP_AUTH_MODE=none` (stdio-only, local). |
494
+ | **Auth scopes** | `tool:<snake_tool_name>:<verb>` or `resource:<kebab-resource-name>:<verb>` (e.g., `tool:inventory_search:read`, `resource:echo-app-ui:read`). Domain-led `<domain>:<verb>` (e.g., `inventory:read`) is an acceptable alternative — pick one convention per server and stay consistent. Skip when no deployment will run `MCP_AUTH_MODE=jwt` or `oauth` — under `none` (stdio, or single-tenant HTTP) scope checks never run. |
477
495
 
478
496
  ### 5. Design Resources
479
497
 
@@ -513,11 +531,12 @@ For services wrapping external APIs, plan the resilience layer.
513
531
  |:--------|:---------|
514
532
  | **Retry boundary** | Service method wraps full pipeline (fetch + parse), not just the network call. Use `withRetry` from `/utils`. |
515
533
  | **Backoff calibration** | Match base delay to upstream recovery time: 200–500ms (ephemeral), 1–2s (rate-limited), 2–5s (degraded). |
516
- | **HTTP status check** | `fetchWithTimeout` already handles this — non-OK → `ServiceUnavailable`. |
534
+ | **HTTP status check** | `fetchWithTimeout` already handles this — a non-2xx throws an `McpError` whose code is mapped from the status (400 → `InvalidParams`, 401 → `Unauthorized`, 403 → `Forbidden`, 404 → `NotFound`, 409 → `Conflict`, 422 → `ValidationError`, 429 → `RateLimited`, 408/504 → `Timeout`, other 5xx → `ServiceUnavailable`; the full table is `api-errors` § *HTTP Response → McpError*), with `status` and `body` on `error.data`, plus `retryAfter` when the upstream sent one. Plan error contracts around those codes, not a blanket `ServiceUnavailable`. |
517
535
  | **Parse failure classification** | Response handler detects HTML error pages and throws transient errors, not `SerializationError`. |
518
536
  | **Exhausted retry messaging** | `withRetry` enriches the final error with attempt count automatically. |
519
- | **Pacing** | No framework primitive paces requests. When the upstream mandates a rate (one request per second, one per five seconds, N concurrent), decide per service how the tool surface honors it — a queue in the service, a concurrency cap on fan-out, or a documented ceiling in the server instructions — and say which. |
520
- | **Caller-supplied URLs or hosts** | Route through `fetchWithTimeout`, which carries the SSRF guard (private ranges, DNS rebinding). Never a bare `fetch` on a caller-controlled destination; `security-pass` audits this sink. |
537
+ | **Total deadline** | Retries multiply a per-attempt timeout: four 30s attempts plus backoff outlast a client's 60s request timeout, and the caller gets a transport timeout instead of this server's classified error. Size `withRetry`'s `deadlineMs` to fit inside the client's timeout and thread `attempt.signal` into each fetch (`api-utils`). |
538
+ | **Pacing** | When the upstream mandates a rate (one request per second, N per minute, N concurrent), put a `createPacer` from `/utils` in front of that service — one pacer per upstream budget, composed as `withRetry` outside and the pacer inside — and note the resulting ceiling in the server instructions when it shapes how an agent should batch work. |
539
+ | **Caller-supplied URLs or hosts** | Route through `fetchWithTimeout` with `rejectPrivateIPs: true` — the SSRF guard is off by default. It blocks private, loopback, link-local, and metadata ranges, and is best-effort: DNS rebinding still gets past it, so a deployment that needs hard isolation adds egress controls. Never a bare `fetch` on a caller-controlled destination; `security-pass` audits this sink. |
521
540
 
522
541
  For API efficiency, design the service methods to minimize upstream calls:
523
542
 
@@ -559,6 +578,7 @@ What this server does, what system it wraps, who it's for.
559
578
 
560
579
  - Bullet list of capabilities and constraints
561
580
  - Auth requirements, rate limits, data access scope
581
+ - Deployment targets (stdio / HTTP / Workers), session mode, upstream credential model, and the terms-of-use constraints from Step 1
562
582
 
563
583
  ## User Goals
564
584
 
@@ -589,13 +609,14 @@ message shape).
589
609
 
590
610
  Draft `instructions` string for `createApp()` — the orientation every client sees at
591
611
  initialize: canonical workflow chain, identifier semantics, rate-limit posture. A short
592
- paragraph; tool descriptions carry the rest.
612
+ paragraph under 2,048 characters, essentials first — Claude Code truncates the string at that
613
+ length, mid-sentence. Tool descriptions carry the rest.
593
614
 
594
615
  ## Implementation Order
595
616
 
596
617
  1. Config and server setup
597
- 2. Services (external API clients)
598
- 3. Reference tool (static, no service dependency — grounds field-testing for everything else)
618
+ 2. Reference tool (static, no service dependency — grounds field-testing for everything else)
619
+ 3. Services (external API clients)
599
620
  4. Read-only tools
600
621
  5. Write tools
601
622
  6. Resources
@@ -611,7 +632,7 @@ Each step is independently testable.
611
632
  ## API Reference <!-- query language, pagination, rate limits; include when worth documenting -->
612
633
  ```
613
634
 
614
- Keep it concise. The design doc is a working reference, not a spec document — enough to orient a developer (or agent) implementing the server, not more.
635
+ Keep it concise. The design doc is a working reference, not a spec document — enough to orient a developer (or agent) implementing the server, not more. It stays true after the build: when the implementation diverges — a renamed field, a dropped method, a changed error code — the doc changes in the same commit, with the why under Design Decisions. The next agent reads it as the spec.
615
636
 
616
637
  **Workflow Analysis example.** For multi-step workflow tools, document the upstream call sequence in a table — it drives several downstream decisions during implementation: the service-layer method shape, retry boundaries, where cleanup or the confirmation round belongs, and what post-action state to fetch for the response.
617
638
 
@@ -644,7 +665,10 @@ Execute the plan using the scaffolding skills:
644
665
  3. `add-resource` for each standalone resource
645
666
  4. `add-prompt` for each prompt
646
667
  5. `add-app-tool` *only if any app tools survived the design step* (rare — see the App Tool row in Step 3)
647
- 6. `devcheck` after each addition
668
+ 6. `add-test` alongside each definition, declared error contracts included
669
+ 7. `devcheck` after each addition
670
+
671
+ Once the surface is built, `tool-defs-analysis` audits the definition language and `field-test` exercises the tools against the live upstream.
648
672
 
649
673
  ## Checklist
650
674
 
@@ -652,8 +676,8 @@ Items without an `If …:` prefix apply to every design. Conditional items only
652
676
 
653
677
  - [ ] Server scope decided — workflow identified, audience sized, boundary drawn (standalone single-API vs. multi-source aggregation vs. internal-only)
654
678
  - [ ] **If multi-source:** tool surface organized around user workflows, not API identity. Sources are service-layer details.
655
- - [ ] External APIs/dependencies researched and verified (docs fetched, SDKs identified)
656
- - [ ] **If wrapping an external API:** live API probed (at minimum: one list/search, one single-item GET, one error case)
679
+ - [ ] External APIs/dependencies researched and verified (docs fetched, SDKs identified, terms of use read — storage, redistribution, AI use, attribution, credential model)
680
+ - [ ] **If wrapping an external API:** live API probed (at minimum: one list/search, one single-item GET, one error case, one unknown-param request)
657
681
  - [ ] User goals enumerated first (3–10 outcomes agents will accomplish, scaled to domain size), then domain operations mapped as raw material
658
682
  - [ ] Each operation classified as tool, resource, prompt, or excluded
659
683
  - [ ] Catastrophically irreversible operations excluded from the tool surface (stay in vendor UI) — not just `destructiveHint`
@@ -672,22 +696,25 @@ Items without an `If …:` prefix apply to every design. Conditional items only
672
696
  - [ ] **If a tool resolves a single identifier:** no-match returns `{ found: false, guidance }` — a result, not a throw — with guidance routing per miss outcome
673
697
  - [ ] **If the domain has opaque vocabulary (codes, identifier formats, coverage windows):** reference tool designed (`topic` enum), implemented first, and used as the routing target in recovery strings and notices
674
698
  - [ ] Annotations set correctly (`readOnlyHint`, `destructiveHint`, `idempotentHint`, `openWorldHint`) — benign writes set `destructiveHint: false` explicitly, read-only tools omit it
675
- - [ ] Server-level `instructions` string drafted — workflow chain, identifier semantics, rate-limit posture (ships via `createApp()` on every initialize)
676
- - [ ] Design doc written to `docs/design.md`
677
- - [ ] Design confirmed with user (or user pre-authorized implementation)
699
+ - [ ] Server-level `instructions` string drafted — workflow chain, identifier semantics, rate-limit posture, under 2,048 characters (ships via `createApp()` on every initialize)
678
700
  - [ ] **If ops share a noun:** related operations consolidated under one tool with a `mode`/`operation` enum — as a `z.discriminatedUnion` input when the arms need different required fields
679
701
  - [ ] **If an upstream API has no native search but the relevant set is bounded:** MCP-side list filtering considered — a distinct local filter param (`filter`/`nameContains`, not `query`), filtering the full set, strict token match (fuzzy only when a caller needs typo tolerance)
680
702
  - [ ] **If the server has workflow tools:** call-flow documented (upstream sequence + mode arms) in design doc's Workflow Analysis
681
703
  - [ ] **If state-aware procedural guidance adds value:** instruction tool considered with `nextToolSuggestions` pre-filled from diagnostics
682
704
  - [ ] **If any tool is config-gated:** nothing routes to it while the gate is off — recovery strings, notices, and `guidance` name a callable target or state the capability is unavailable, and structured follow-ups naming it are emitted only under the config that registers it
683
705
  - [ ] **If workflow tools have destructive modes:** destructive arm gated on a `ctx.requestInput` confirmation read back from `ctx.inputs`, with `destructiveHint` annotation so clients that never fulfil the round still surface the risk
706
+ - [ ] **If any tool calls `ctx.requestInput`:** `createApp()` declares `sessionMode` with `require: 'stateful'`
707
+ - [ ] **If any output carries text other people wrote:** those fields listed, `format()` quotes or fences free text and flattens CR/LF in inline slots, and the server instructions say the content is data
684
708
  - [ ] **If a parameter determines blast radius:** safe default set (e.g., `mode: 'preview'`, `dryRun: true`, `confirmCount` required)
685
709
  - [ ] **If an action's effect is observable only after the call (wake, trigger, dispatch, provision):** confirmation on by default through one numeric window param (`0` = act and return), pre-probe then poll with early return, every outcome a result rather than a throw, and a read-only sibling tool that probes the same state
686
710
  - [ ] **App tools default to no.** If one was proposed, verified there's a real human-in-the-loop in an MCP Apps-capable client justifying the iframe/CSP/`format()`-twin maintenance cost — otherwise dropped in favor of a standard tool
687
711
  - [ ] **If the server exposes resources:** URIs use `{param}` templates, pagination planned for large lists
688
712
  - [ ] **If the server is itself the source of truth (no external API):** state lifecycle planned — tenant-scoped vs. global, TTLs, what survives restart, storage backend chosen
689
713
  - [ ] **If the server has external deps or shared state:** service layer planned (or explicitly skipped with reasoning)
690
- - [ ] **If services wrap external APIs:** resilience planned (retry boundary, backoff, parse classification)
691
- - [ ] **If multi-source server:** each source has its own service with independent auth/retry/rate-limit config. Fallback chains or fan-out strategy documented per tool. Output includes source provenance.
714
+ - [ ] **If services wrap external APIs:** resilience planned (retry boundary, backoff, parse classification, a total deadline inside the client timeout, a pacer where the upstream mandates a rate, `rejectPrivateIPs` on caller-supplied URLs)
715
+ - [ ] **If multi-source server:** each source has its own service with independent auth/retry/rate-limit config. Fallback chains or fan-out strategy documented per tool. Output includes source provenance. An input rejection from any source fails the call rather than reading as that source's outage.
716
+ - [ ] **If mirroring a bulk upstream:** the terms permit storing the data, and a local index reproduces the upstream's search — no server-side ranking or query expansion on the mirrored path
692
717
  - [ ] **If exposing a SQL/analytical workspace is in scope:** DataCanvas considered (`api-canvas` skill), and it earns its keep on *analytical* fit (an agent would SQL it), not row count — a discovery/search surface of categorical metadata doesn't qualify. Any tool emitting a `canvas_id` is paired with a `dataframe_query` (+ `dataframe_describe`) tool in the same surface — a token with no query tool is dead output
693
718
  - [ ] **If the server needs runtime config:** env vars identified in `server-config.ts`
719
+ - [ ] Design doc written to `docs/design.md` — every example value synthetic, no keys, tokens, or private hosts
720
+ - [ ] Design confirmed with user (or user pre-authorized implementation)
@@ -4,7 +4,7 @@ description: >
4
4
  Land working-tree changes as logical commits — the work grouped by concern, topped by a release commit (version bump, changelog, regenerated artifacts). The work commits land first, then the version bump, verification, and the release commit on top. Stops at "committed locally on main" — or, when the project releases through a release PR, at "release branch pushed, PR open". No tag, no push to main, no publish: the release-and-publish skill merges, tags, and ships from here. Distilled from the git_wrapup_instructions protocol.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "1.19"
7
+ version: "1.25"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -28,7 +28,7 @@ A project can route every release through a pull request — one PR per version,
28
28
  | **gated** | commit stack on `release/<version>`, branch pushed, PR open | a review pass on the PR (`release-pr-review` skill), then a separate `release-and-publish` run fast-forwards `main`, tags, and ships |
29
29
  | **straight-through** | same as gated | the same agent continues straight into `release-and-publish` |
30
30
 
31
- The branch is created at wrapup time, never before: work happens on `main` until the version is known, then the uncommitted tree moves to `release/<version>` in one step (step 3), ahead of the first commit. The commit stack, the release commit, and the tag format are identical in every mode — the PR adds an artifact around them, it does not change them.
31
+ The branch is created at wrapup time, never before — its name carries the version, so it cannot exist until step 2 has settled one. The moment that number is known, the uncommitted tree moves to `release/<version>` (end of step 2), and every commit in the run lands there. **No commit in these two modes ever reaches `main`, including the first one:** a stack committed on `main` and then branched is a rewrite to undo, and once it is pushed there is no undo, because force-push is banned. The commit stack, the release commit, and the tag format are identical in every mode — the PR adds an artifact around them, it does not change them.
32
32
 
33
33
  ## Pre-wrapup gate checklist
34
34
 
@@ -37,7 +37,7 @@ Every item must be true before starting wrapup. Committing means releasing — a
37
37
  - [ ] **Changes exist** — uncommitted files or commits since the last tag
38
38
  - [ ] **Work is complete** — no half-finished features, no "I'll add the test later," no TODO placeholders. The diff represents a shippable unit.
39
39
  - [ ] **Code simplified** — if the diff spans more than ~50 changed lines or touches 3+ source files, the `code-simplifier` skill has been run across the changes
40
- - [ ] **`bun run devcheck` passes** — typecheck + lint clean
40
+ - [ ] **`bun run devcheck` is clean — exit-0 AND zero warnings.** Biome/lint warnings exit 0 (non-blocking to the tool) but are a HARD BLOCK to shipping — pre-existing warnings in untouched code included: fix them behavior-preserving, or get the maintainer's explicit waiver in the maintainer's own words — never your own adjudication. A documented alternative remedy in a skill (`lint:mcp`'s "or verify by hand that `format()` renders it") is a way to UNDERSTAND a warning, never a licence to ship it; hand-verifying it and filing a follow-up issue is still shipping dirty. Nor is a pinned version a reason to defer the real fix: when the correct fix crosses the minor floor, the VERSION yields, not the gate — take the minor and say so. Two things that look dirty but aren't: (1) **`info` is not `warning`** — Biome's unsafe-fix suggestions ("Skipped N suggested fixes", `useLiteralKeys` and friends) print at info severity with the step still ✅; not the block. (2) **The Security Audit step's transitive warn** — a DIRECT-dep advisory is a hard failure, an all-transitive one is `⚠️ WARNING` with the run still ✅; confirm from the `bun audit` dependency path (`pkg › child` = transitive), then ship it. Never force a gate green with a `resolutions`/`overrides` hack, and never run `audit:refresh` to silence it (it re-resolves the `^`-ranged framework pin off its hold). The exception is a deliberate, maintainer-directed `overrides` block as its own dependency-hygiene pass — mcp-ts-core carries one as of 0.11.0: leave it in place, pre-authorize it by name in implement/release briefs (or agents burn a phase chasing it), pin entries to the lowest patched version *inside the consumer's existing major*, and verify the installed tree rather than assuming a pin applied (bun skips resolutions its range can't satisfy). Read the severity label and the failing step before calling a green run dirty — filtering devcheck's output strips exactly the context that separates the tiers.
41
41
  - [ ] **`bun run rebuild` succeeds** — full clean build from scratch
42
42
  - [ ] **All tests pass** — `bun run test:all` (or `bun run test`), plus `bun run test:package` where the project defines one: it guards the public-export manifest and is not part of `test:all`. New tests and regression tests added as needed for the changes being shipped.
43
43
  - [ ] **Fixes verified** — bug fixes validated, generally via `bun run rebuild` and field-testing. Not just written — confirmed to resolve the described behavior.
@@ -64,7 +64,7 @@ Diff against `HEAD`, not the index: plain `git diff` omits staged changes entire
64
64
 
65
65
  If the working tree is clean AND there are no commits since the last tag, halt — nothing to wrap up.
66
66
 
67
- ### 2. Determine the new version
67
+ ### 2. Determine the new version — and, in release PR mode, create its branch
68
68
 
69
69
  Read the current version from `package.json`. Apply the intended bump:
70
70
 
@@ -76,16 +76,18 @@ Read the current version from `package.json`. Apply the intended bump:
76
76
 
77
77
  Default to **patch** unless the diff clearly warrants minor or major.
78
78
 
79
- ### 3. Commit the work — one commit per concern
80
-
81
- **Release PR mode only — move to the release branch first, before the first commit:**
79
+ **In `gated` or `straight-through` mode, create the branch now, before going on to step 3** — the version you just settled is its name, and step 3 opens by committing:
82
80
 
83
81
  ```bash
84
82
  git branch --show-current # must be main
85
83
  git switch -c release/<version> # uncommitted work rides along
86
84
  ```
87
85
 
88
- Commits never land on `main` in this mode. If a `release/*` branch already exists locally, a prior release PR was never merged — halt and report it rather than stacking a second release on top.
86
+ Do not defer this to "just before the first commit". Step 3 is where committing starts, so a branch not created here is a branch created too late. If `git branch --no-merged main --list 'release/*'` prints a branch, a prior release PR was never merged — halt and report it rather than stacking a second release on top. A `release/*` branch already merged into `main` is a leftover from a finished release, not a blocker: delete it with `git branch -d` and continue.
87
+
88
+ ### 3. Commit the work — one commit per concern
89
+
90
+ **Release PR mode: `git branch --show-current` must print `release/<version>` before you run the first `git commit`.** If it prints `main`, step 2's branch step was skipped — go back and do it. The uncommitted tree moves with you, so nothing is lost by branching late, but a commit already on `main` has to be unwound.
89
91
 
90
92
  **The work is committed before the version is bumped.** Work concerns routinely share a file with the version — a dependency refresh edits `package.json`, a doc edit lands in a `CLAUDE.md`/`AGENTS.md` that pins a version string — and the file is the atomic boundary, so whichever commit comes first takes the file whole. Committing the work first leaves the version hunk (step 4) as the only thing those files carry into the release commit.
91
93
 
@@ -103,6 +105,8 @@ git commit --only <paths-for-this-concern> -m "<subject>" -m "<body>"
103
105
 
104
106
  **The file is the atomic boundary:** NEVER split a single file's working-tree changes across commits, regardless of mechanism — not `git add -p`, not an index-only patch (`git apply --cached`), not editing the file between commits to remove-then-re-add a hunk. When one file serves two concerns, it ships whole in the commit of its dominant concern; a later commit may touch the file again only for changes made AFTER the first commit (the version bump applied in step 4).
105
107
 
108
+ **Every commit builds on its own.** When a concern changes an exported contract — a service method's return type, a shared helper's signature — the files that consume it ride in the same commit, even when they also carry other concerns. Grouping the contract change into one commit and each consumer into its own later commit leaves pushed commits that fail typecheck alone, and pushed history is never rewritten to repair them.
109
+
106
110
  **Subject format:** Conventional Commits, no version in the subject — `feat: hosted server endpoint`, `fix: handle empty SPARQL result sets`, `feat(linter): enrichment contract rules`, `docs: document the enrichment block`, `chore(deps): refresh dev dependencies`.
107
111
 
108
112
  **Body: every commit has one, and it is one or two lines.** Uniform across the stack — no commit ships subject-only, none ships a paragraph. One sentence stating the *why* or the load-bearing constraint, a second only if the first genuinely cannot carry it. Two lines is the hard ceiling.
@@ -175,7 +179,7 @@ security: false # true ONLY for a security fix in this server's own source
175
179
 
176
180
  **Tone:** Terse, fact-dense. Bullet = **symbol** + what changed + at most one consumer-facing caveat; one sentence by default, two max — a bullet past ~40 words or three sentences is wrong. The linked issue carries the why and the commit diff the how; the changelog names what changed and what a consumer does about it. Cut: history/justification narration, design-rationale defense, "X unchanged" clauses (short parenthetical only where a misread is likely), edge-case inventories. **Verified ≠ included** — the diff-is-source-of-truth rule bounds the truth of what you write, never the amount. Model length on `changelog/template.md`'s authoring guide, never on the previous entry (entries modeled on entries compound). `agent-notes` carries adoption steps only, never a second rendering of the body; a consequence shared by many bullets is stated once, not per bullet. Full conventions: the authoring guide in `changelog/template.md`.
177
181
 
178
- **Re-read the entry file after writing it, then sweep for harness markup:** `grep -rlF -e '</invoke>' -e '</content>' changelog/` must print nothing. A stray closing tag at EOF is the authoring tool's own syntax bleeding into the file; `changelog/` is in `package.json` `files`, so it ships inside the npm tarball, and `changelog:check` cannot catch it — the rollup drops the trailing line, so a clean `CHANGELOG.md` proves nothing about the entry.
182
+ **Re-read the entry file after writing it, then sweep for harness markup:** `grep -rlF -e '</invoke>' -e '</content>' changelog/` must print nothing. The sweep covers the whole directory: a hit in an older entry gets deleted and ships in this release's commit, never left as out of scope, because every entry ships in every tarball. A stray closing tag at EOF is the authoring tool's own syntax bleeding into the file; `changelog/` is in `package.json` `files`, so it ships inside the npm tarball, and `changelog:check` cannot catch it — the rollup drops the trailing line, so a clean `CHANGELOG.md` proves nothing about the entry.
179
183
 
180
184
  ### 6. Regenerate derived artifacts
181
185
 
@@ -221,11 +225,13 @@ Skip this step entirely when the project has no release PR mode — go to step 1
221
225
 
222
226
  ```bash
223
227
  git push -u origin release/<version>
224
- gh pr create --base main --head release/<version> --title "<release commit subject>" --body-file <path-to-body.md>
228
+ gh pr create --base main --head release/<version> --title "<release commit subject>" --assignee @me --body-file <path-to-body.md>
225
229
  ```
226
230
 
227
231
  **Title:** the release commit's subject, verbatim — `chore(release): <version> — <theme>`.
228
232
 
233
+ **Assignee:** `--assignee @me` on every release PR, so it lands in the maintainer's assigned queue like a filed issue. A reviewer is not set: GitHub drops a review request aimed at the PR's own author and refuses a self-approval, so on a self-authored release PR the reviewer field stays empty by design — the review record is the summary comment `release-pr-review` leaves.
234
+
229
235
  **Body — always via `--body-file`, never an inline `--body` string** (backticks inside a double-quoted argument are command substitution and silently vanish). Write the file to a scratch location, not into the repo.
230
236
 
231
237
  The body is the release digest — the `## Changes` bullets and changelog link the annotated tag will carry, under a theme line and above a gates record that both stay on the PR. The digest is written here, reviewed on the PR, and copied into the tag at release time, so it is the one place the release notes get reviewed before they become permanent. Format:
@@ -4,7 +4,7 @@ description: >
4
4
  Investigate, adopt, and verify dependency updates — with special handling for `@cyanheads/mcp-ts-core`. Captures what changed, understands why, cross-references against the codebase, adopts framework improvements, syncs project skills, and runs final checks. Supports two entry modes: run the full flow end-to-end, or review updates you already applied.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "2.8"
7
+ version: "2.9"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -85,7 +85,7 @@ Cross-reference each finding against the server's code. Collect adoption opportu
85
85
 
86
86
  **Template review.** The framework also ships `templates/CLAUDE.md` and `templates/AGENTS.md` as scaffolding for consumer agent protocol files. The consumer's `CLAUDE.md`/`AGENTS.md` was copied at init time and has since diverged (local customizations, echo replacements, server-specific sections). Read the upstream template fresh at `node_modules/@cyanheads/mcp-ts-core/templates/CLAUDE.md`.
87
87
 
88
- Read the upstream template end-to-end, mentally comparing against the current `CLAUDE.md`/`AGENTS.md`. Apply framework-authored updates directly — new skill references in the skills table, new entries in the "What's Next?" section, updated convention callouts, clarified patterns. These are factual updates, not taste decisions; the consumer's agent protocol file is meant to track the framework's. Only surface a decision when a template change conflicts with a section the consumer has intentionally customized — a section is "intentionally customized" when it contains server-specific domain context, bespoke checklists, or content that doesn't originate from the template. In that case, note the conflict and ask.
88
+ Read the upstream template end-to-end and compare it against the current `CLAUDE.md`/`AGENTS.md` section by section — the skills table row by row, since a row whose wording changed (a skill's described behavior) drifts as silently as a missing one. Apply framework-authored updates directly — new or reworded skill references in the skills table, new entries in the "What's Next?" section, updated convention callouts, clarified patterns. These are factual updates, not taste decisions; the consumer's agent protocol file is meant to track the framework's. Only surface a decision when a template change conflicts with a section the consumer has intentionally customized — a section is "intentionally customized" when it contains server-specific domain context, bespoke checklists, or content that doesn't originate from the template. In that case, note the conflict and ask.
89
89
 
90
90
  ### 5. Sync project skills and scripts
91
91
 
@@ -4,7 +4,7 @@ description: >
4
4
  Pick and run a multi-phase workflow that chains foundational task skills (`git-wrapup`, `release-and-publish`, `maintenance`, `field-test`, `setup`, etc.) end-to-end. Routes user intent to a workflow file under `workflows/` — greenfield builds, maintenance + release, field-test + fix, or known-work + release. Single source for the universal rules (no commits without authorization, no destructive git, no marketing language), the orchestrator posture (own the goal, ground sub-agents in primary sources, verify against the goal), and the sub-agent strategy (orient block, parallel fanout, isolation, normalization) that apply across every workflow. Sub-agents are an optional capability — workflows run linearly when fanout isn't available.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "1.10"
7
+ version: "1.11"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -4,7 +4,7 @@ description: >
4
4
  Workflow: scaffold one or more new MCP server projects from `bunx @cyanheads/mcp-ts-core init` through design → build → polish → first public release. Each phase invokes a foundational skill end-to-end; this file is the sequencing and gates, not the procedural detail. Read `../SKILL.md` first for the universal rules and sub-agent strategy.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "1.1"
7
+ version: "1.2"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -36,7 +36,7 @@ Everything stays at **v0.1.0** through the build. Intermediate commits don't bum
36
36
  | Phase | Tier 1 skill(s) |
37
37
  |:---|:---|
38
38
  | Scaffold (1) | `framework-skills/setup/SKILL.md` |
39
- | Initial commit, design commit, build commit, pre-launch commit (2, 5, 10, 16) | `framework-skills/git-wrapup/SKILL.md` (commit + tag, no push) |
39
+ | Initial commit, design commit, build commit, pre-launch commit (2, 5, 10, 16) | `framework-skills/git-wrapup/SKILL.md` step 3 commit conventions only — see "Checkpoint commits" below |
40
40
  | Design + validation (3, 4) | `framework-skills/design-mcp-server/SKILL.md` |
41
41
  | Build (6) | `framework-skills/add-tool/SKILL.md`, `framework-skills/add-app-tool/SKILL.md`, `framework-skills/add-resource/SKILL.md`, `framework-skills/add-prompt/SKILL.md`, `framework-skills/add-service/SKILL.md` |
42
42
  | Tool-def audit (7) | `framework-skills/tool-defs-analysis/SKILL.md` |
@@ -70,8 +70,8 @@ Each phase's Objective column is the goal state per target — the verifiable en
70
70
  | 14 | Security pass | `security-pass` findings addressed; no open security gaps | parallel fanout | gate-free |
71
71
  | 15 | Final-state check | `rebuild` + `devcheck` + `test:all` + `lint:packaging` green; LICENSE present; no unfinished TODO/FIXME | orchestrator-direct | gate-free |
72
72
  | 16 | Pre-launch commit | Final polish + security work committed and pushed | parallel fanout | **barrier** — human decision: version-bump intent (typically v0.1.1) |
73
- | 17 | Final wrap-up | Launch version (typically v0.1.1) commit + annotated tag in place; **not pushed** | parallel fanout (Bash git only) | **barrier** — release authorization required before push and publish |
74
- | 18 | Release | Pushed and published per scope; tag annotation renders as structured markdown on GitHub Release; artifacts reachable | parallel fanout or serial (per npm 2FA mode) | — |
73
+ | 17 | Final wrap-up | Launch version (typically v0.1.1) release commit on top of the stack — on `main`, or on a pushed `release/<version>` branch with the PR open in release PR mode; no tag | parallel fanout (Bash git only) | **barrier** — release authorization required before push and publish |
74
+ | 18 | Release | Repo public when the release is public; merged (release PR mode), tagged, pushed, and published per scope; tag annotation renders as structured markdown on GitHub Release; artifacts reachable | parallel fanout or serial (per npm 2FA mode) | — |
75
75
 
76
76
  Phase 11 is optional. Phase 12 is the last phase that modifies source code — everything after is docs/metadata/verification.
77
77
 
@@ -85,6 +85,9 @@ Sub-agent runs `bunx @cyanheads/mcp-ts-core init <name>`, follows the `setup` sk
85
85
  ### Phase 2: Initial commit
86
86
  Sub-agent verifies `gh repo view --json visibility` returns `PRIVATE` (or has explicit user authorization for public) before push. Tag is `v0.1.0`.
87
87
 
88
+ ### Checkpoint commits (Phases 2, 5, 10, 16)
89
+ Plain commits on `main`, pushed to the private repo. They follow `git-wrapup`'s step 3 conventions — grouped by concern, staged and committed by pathspec, one- or two-line bodies — and nothing else from that skill: no version bump, no changelog entry, no release branch or PR. Run end to end, `git-wrapup` bumps the version and, when the project declares a release PR mode, moves the work to `release/<version>` and opens a PR; that belongs to Phase 17 alone. Only Phase 2 tags (`v0.1.0`, annotated, `--cleanup=whitespace`).
90
+
88
91
  ### Phase 4: Design validation
89
92
  Two sub-agents per target, sequential:
90
93
 
@@ -100,7 +103,7 @@ Sub-agents will exhaust context on targets with 4+ tools — work persists to di
100
103
  For each tool / resource / prompt named in `docs/design.md`, verify a definition file exists in `src/mcp-server/{tools,resources,prompts}/definitions/`. For missing surface, decide: implement it (spawn a narrow-scope sub-agent), drop it from the design (update `docs/design.md`), or defer to a follow-up (record in the Decisions Log). This is orchestration glue — small enough that the orchestrator can run it directly for N ≤ 3, fan out for larger N.
101
104
 
102
105
  ### Phase 11: Field-test loop (optional)
103
- When the upstream API supports live testing and an API key is available, run the phases of `field-test-fix.md` as a sub-loop here, ending at its field-test commit. Skip with a note if blocked.
106
+ When the upstream API supports live testing and an API key is available, run Phases 1–5 of `field-test-fix.md` as a sub-loop here (field-test → triage → fix → verify → loop decision). Skip its Phase 6 wrap-up + release: the fixes stay in the working tree and land in the next checkpoint commit. Its Phase 7 issue cleanup runs after the Phase 18 launch, since nothing else closes the issues the loop filed. Skip with a note if blocked.
104
107
 
105
108
  ### Phase 12: Simplify
106
109
  Last phase that modifies source code. Everything after is docs/metadata/verification.
@@ -109,7 +112,10 @@ Last phase that modifies source code. Everything after is docs/metadata/verifica
109
112
  Orchestrator-direct mechanical verification per target: `bun run rebuild`, `bun run devcheck`, `bun run test:all` (or `test`), `bun run lint:packaging`. `LICENSE` present. No `TODO`/`FIXME` indicating unfinished work. `CHANGELOG.md` current. `docs/tree.md` reflects current structure. Fix anything red before Phase 16; this is verification, not a sub-agent task.
110
113
 
111
114
  ### Phase 17: Final wrap-up
112
- Version bump intent is typically **patch** — v0.1.0 was the scaffold tag; the launch is the first real release at v0.1.1. Bash git only; **do not push** — Phase 18 owns the push.
115
+ Version bump intent is typically **patch** — v0.1.0 was the scaffold tag; the launch is the first real release at v0.1.1. Runs `git-wrapup` end to end, Bash git only. In release PR mode it pushes `release/<version>` and opens the PR; otherwise nothing is pushed. No tag — Phase 18 merges, tags, pushes `main`, and publishes.
116
+
117
+ ### Phase 18: Release
118
+ `release-and-publish` never changes repo visibility. When the release is public, the orchestrator makes the repo public before the release runs: scan the full git history (not just tracked files) for secrets and private content, since every commit goes public, then `gh repo edit <owner>/<repo> --visibility public --accept-visibility-change-consequences`. Publishing from a still-private repo leaves the npm repository link, the GitHub Release, and the `.mcpb` download URL unreachable.
113
119
 
114
120
  ## Workflow-specific gotchas
115
121
 
@@ -119,6 +125,7 @@ Version bump intent is typically **patch** — v0.1.0 was the scaffold tag; the
119
125
  | 2 | Build sub-agents exhaust context on targets with 4+ tools | Expected — plan a finish iteration with a concrete punch list, narrow scope |
120
126
  | 3 | Design gate sub-agents flag style preferences as failures | Gate prompt: "Do NOT flag style preferences or marginal scope suggestions — only structural issues that would cause wasted build effort" |
121
127
  | 4 | Sub-agent commits during Phase 1 despite the orchestration override | Phase 1 prompt restates: "Do NOT commit — leave working tree dirty for Phase 2" verbatim |
128
+ | 5 | A checkpoint commit routed through `git-wrapup` end to end bumps the version mid-build, or opens a release PR in release PR mode | Checkpoint commits use `git-wrapup`'s commit conventions only (see "Checkpoint commits"); the full skill runs once, in Phase 17 |
122
129
 
123
130
  ## Checklist
124
131
 
@@ -139,5 +146,5 @@ Version bump intent is typically **patch** — v0.1.0 was the scaffold tag; the
139
146
  - [ ] Phase 14: security-pass complete, findings addressed
140
147
  - [ ] Phase 15: final-state check — rebuild + devcheck + test:all + lint:packaging green; LICENSE; no TODO/FIXME
141
148
  - [ ] Phase 16: pre-launch commit per target
142
- - [ ] Phase 17: final wrap-up — version bumped, changelog authored, commit + annotated tag per target
143
- - [ ] Phase 18: release — published per scope, artifacts verified reachable
149
+ - [ ] Phase 17: final wrap-up — version bumped, changelog authored, release commit per target (release PR open in release PR mode); no tag
150
+ - [ ] Phase 18: release — repo public first when the release is public (full-history scan clean), published per scope, artifacts verified reachable; field-test issues closed with the version that fixed them
@@ -4,7 +4,7 @@ description: >
4
4
  Finalize documentation and project metadata for a ship-ready MCP server. Use after implementation is complete, tests pass, and devcheck is clean. Safe to run at any stage — each step checks current state and only acts on what still needs work.
5
5
  metadata:
6
6
  author: cyanheads
7
- version: "2.17"
7
+ version: "2.18"
8
8
  audience: external
9
9
  type: workflow
10
10
  ---
@@ -211,7 +211,7 @@ If the project ships as an `.mcpb` bundle for Claude Desktop (check for `manifes
211
211
  - `manifest.json` version matches `package.json` version
212
212
  - Env var names in `manifest.json` (`mcp_config.env` + `user_config`) match `server.json` `environmentVariables` — `lint:packaging` enforces this, but verify the set is complete
213
213
  - `manifest.json` `name` matches `package.json` name **without the npm scope prefix** (e.g. `bls-mcp-server`, not `@cyanheads/bls-mcp-server`); `description` matches `package.json`
214
- - `manifest.json` `author` is the full person object — `{ "name", "email", "url" }` — carrying the same identity as `package.json` `author` (name matches the LICENSE copyright holder, url is the author's site)
214
+ - `manifest.json` `author` is `{ "name": "<publisher handle>" }` — the same handle as the `.claude-plugin` / `.codex-plugin` `author.name` and the GitHub owner (e.g. `{ "name": "cyanheads" }`), not the LICENSE copyright holder's person object; `package.json` `author` is where the full `Name <email> (url)` identity lives
215
215
  - `manifest.json` `user_config` entries must include `title` and `type` fields — `mcpb pack` validates these
216
216
  - Every `user_config` entry is referenced from `mcp_config.env` as `"X": "${user_config.X}"`, and `mcp_config` carries no other `${…}` besides MCPB's own path placeholders (`${__dirname}`, `${HOME}`, …). The host substitutes nothing else: a declared option that is never referenced is collected and dropped, and `"X": "${X}"` reaches the server as that literal string. `lint:packaging` enforces both
217
217
  - For each `user_config` entry referenced as `${user_config.X}` in `mcp_config.env`: if it's not `required: true`, set `"default": ""`. MCPB hosts (Claude Desktop included) pass the literal placeholder string through to the process when an optional field is left blank without a default — the `default` keeps that string out of the process. Server-side, the framework already treats a whole-value `${…}` placeholder the same as an empty string — unset — in both its own config and `parseEnvConfig`, so an optional field falls through to its default and a required one fails as missing rather than as a format error; a per-field `z.preprocess` guard for placeholders is redundant and can be dropped.
@@ -28,7 +28,7 @@ These are set by `init` and generally don't need changes. Verify they're present
28
28
  | `types` | `"dist/index.d.ts"` | TypeScript declarations |
29
29
  | `files` | `["dist/"]` | What npm publishes |
30
30
  | `engines` | `{ "node": ">=24.0.0", "bun": ">=1.4.0" }` | Node runs the built `dist/`; Bun is the dev floor |
31
- | `packageManager` | `"bun@1.4.0"` | Pins the dev package manager; keep current with the framework's Bun version |
31
+ | `packageManager` | `"bun@1.4.2"` | Pins the dev package manager; keep current with the framework's Bun version |
32
32
  | `scripts` | _(various)_ | Build, dev, test scripts |
33
33
  | `dependencies` | `@cyanheads/mcp-ts-core` | Core framework |
34
34