@cyanheads/mcp-ts-core 0.13.6 → 0.13.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +3 -3
- package/CLAUDE.md +3 -3
- package/README.md +55 -52
- package/biome.json +1 -1
- package/changelog/0.13.x/0.13.7.md +77 -0
- package/config/tsconfig.base.json +2 -2
- package/dist/config/index.d.ts.map +1 -1
- package/dist/config/index.js +42 -11
- package/dist/config/index.js.map +1 -1
- package/dist/core/app.d.ts.map +1 -1
- package/dist/core/app.js +21 -4
- package/dist/core/app.js.map +1 -1
- package/dist/core/context.d.ts +9 -1
- package/dist/core/context.d.ts.map +1 -1
- package/dist/core/context.js +4 -13
- package/dist/core/context.js.map +1 -1
- package/dist/core/worker.d.ts.map +1 -1
- package/dist/core/worker.js +7 -1
- package/dist/core/worker.js.map +1 -1
- package/dist/mcp-server/handlerContext.d.ts +6 -0
- package/dist/mcp-server/handlerContext.d.ts.map +1 -1
- package/dist/mcp-server/handlerContext.js +3 -0
- package/dist/mcp-server/handlerContext.js.map +1 -1
- package/dist/mcp-server/prompts/prompt-registration.d.ts.map +1 -1
- package/dist/mcp-server/prompts/prompt-registration.js +6 -3
- package/dist/mcp-server/prompts/prompt-registration.js.map +1 -1
- package/dist/mcp-server/tools/utils/toolHandlerFactory.d.ts.map +1 -1
- package/dist/mcp-server/tools/utils/toolHandlerFactory.js +5 -1
- package/dist/mcp-server/tools/utils/toolHandlerFactory.js.map +1 -1
- package/dist/mcp-server/transports/http/httpErrorHandler.d.ts.map +1 -1
- package/dist/mcp-server/transports/http/httpErrorHandler.js +15 -5
- package/dist/mcp-server/transports/http/httpErrorHandler.js.map +1 -1
- package/dist/storage/core/IStorageProvider.d.ts +5 -2
- package/dist/storage/core/IStorageProvider.d.ts.map +1 -1
- package/dist/storage/core/providerHelpers.d.ts +29 -8
- package/dist/storage/core/providerHelpers.d.ts.map +1 -1
- package/dist/storage/core/providerHelpers.js +49 -11
- package/dist/storage/core/providerHelpers.js.map +1 -1
- package/dist/storage/providers/cloudflare/d1Provider.js +4 -4
- package/dist/storage/providers/cloudflare/d1Provider.js.map +1 -1
- package/dist/storage/providers/cloudflare/kvProvider.d.ts +2 -0
- package/dist/storage/providers/cloudflare/kvProvider.d.ts.map +1 -1
- package/dist/storage/providers/cloudflare/kvProvider.js +11 -9
- package/dist/storage/providers/cloudflare/kvProvider.js.map +1 -1
- package/dist/storage/providers/cloudflare/r2Provider.d.ts.map +1 -1
- package/dist/storage/providers/cloudflare/r2Provider.js +8 -5
- package/dist/storage/providers/cloudflare/r2Provider.js.map +1 -1
- package/dist/storage/providers/fileSystem/fileSystemProvider.d.ts +1 -0
- package/dist/storage/providers/fileSystem/fileSystemProvider.d.ts.map +1 -1
- package/dist/storage/providers/fileSystem/fileSystemProvider.js +10 -8
- package/dist/storage/providers/fileSystem/fileSystemProvider.js.map +1 -1
- package/dist/storage/providers/inMemory/inMemoryProvider.d.ts +5 -0
- package/dist/storage/providers/inMemory/inMemoryProvider.d.ts.map +1 -1
- package/dist/storage/providers/inMemory/inMemoryProvider.js +9 -5
- package/dist/storage/providers/inMemory/inMemoryProvider.js.map +1 -1
- package/dist/storage/providers/supabase/supabaseProvider.d.ts.map +1 -1
- package/dist/storage/providers/supabase/supabaseProvider.js +5 -1
- package/dist/storage/providers/supabase/supabaseProvider.js.map +1 -1
- package/dist/testing/index.d.ts +6 -4
- package/dist/testing/index.d.ts.map +1 -1
- package/dist/testing/index.js +6 -4
- package/dist/testing/index.js.map +1 -1
- package/dist/utils/formatting/partialResult.d.ts +28 -2
- package/dist/utils/formatting/partialResult.d.ts.map +1 -1
- package/dist/utils/formatting/partialResult.js +46 -2
- package/dist/utils/formatting/partialResult.js.map +1 -1
- package/dist/utils/internal/error-handler/errorHandler.d.ts +8 -2
- package/dist/utils/internal/error-handler/errorHandler.d.ts.map +1 -1
- package/dist/utils/internal/error-handler/errorHandler.js +27 -16
- package/dist/utils/internal/error-handler/errorHandler.js.map +1 -1
- package/dist/utils/internal/error-handler/mappings.d.ts +1 -0
- package/dist/utils/internal/error-handler/mappings.d.ts.map +1 -1
- package/dist/utils/internal/error-handler/mappings.js +1 -0
- package/dist/utils/internal/error-handler/mappings.js.map +1 -1
- package/dist/utils/internal/performance.d.ts +5 -1
- package/dist/utils/internal/performance.d.ts.map +1 -1
- package/dist/utils/internal/performance.js +13 -8
- package/dist/utils/internal/performance.js.map +1 -1
- package/dist/utils/security/sanitization.d.ts +15 -15
- package/dist/utils/security/sanitization.d.ts.map +1 -1
- package/dist/utils/security/sanitization.js +108 -88
- package/dist/utils/security/sanitization.js.map +1 -1
- package/dist/utils/telemetry/instrumentation.d.ts +6 -2
- package/dist/utils/telemetry/instrumentation.d.ts.map +1 -1
- package/dist/utils/telemetry/instrumentation.js +23 -8
- package/dist/utils/telemetry/instrumentation.js.map +1 -1
- package/framework-skills/add-app-tool/SKILL.md +12 -18
- package/framework-skills/add-prompt/SKILL.md +3 -1
- package/framework-skills/add-provider/SKILL.md +14 -4
- package/framework-skills/add-resource/SKILL.md +3 -3
- package/framework-skills/add-tool/SKILL.md +22 -7
- package/framework-skills/api-canvas/SKILL.md +2 -2
- package/framework-skills/api-config/SKILL.md +4 -3
- package/framework-skills/api-context/SKILL.md +8 -5
- package/framework-skills/api-errors/SKILL.md +7 -3
- package/framework-skills/api-linter/SKILL.md +8 -8
- package/framework-skills/api-telemetry/SKILL.md +9 -4
- package/framework-skills/api-testing/SKILL.md +21 -13
- package/framework-skills/api-utils/SKILL.md +3 -3
- package/framework-skills/api-utils/references/security.md +7 -6
- package/framework-skills/code-simplifier/SKILL.md +31 -18
- package/framework-skills/design-mcp-server/SKILL.md +62 -35
- package/framework-skills/git-wrapup/SKILL.md +16 -10
- package/framework-skills/maintenance/SKILL.md +2 -2
- package/framework-skills/orchestrations/SKILL.md +1 -1
- package/framework-skills/orchestrations/workflows/greenfield-build.md +15 -8
- package/framework-skills/polish-docs-meta/SKILL.md +2 -2
- package/framework-skills/polish-docs-meta/references/package-meta.md +1 -1
- package/framework-skills/polish-docs-meta/references/readme.md +3 -3
- package/framework-skills/release-and-publish/SKILL.md +6 -4
- package/framework-skills/release-pr-review/SKILL.md +18 -1
- package/framework-skills/report-issue-framework/SKILL.md +2 -2
- package/framework-skills/report-issue-local/SKILL.md +3 -3
- package/framework-skills/security-pass/SKILL.md +11 -3
- package/framework-skills/tool-defs-analysis/SKILL.md +3 -3
- package/package.json +15 -36
- package/templates/.env.example +3 -1
- package/templates/.github/ISSUE_TEMPLATE/feature_request.yml +1 -1
- package/templates/AGENTS.md +2 -2
- package/templates/CLAUDE.md +2 -2
- package/templates/Dockerfile +4 -4
- package/templates/package.json +3 -3
- package/templates/src/mcp-server/prompts/definitions/echo.prompt.ts +2 -4
- package/templates/src/mcp-server/resources/definitions/echo-app-ui.app-resource.ts +51 -14
- package/templates/src/mcp-server/resources/definitions/echo.resource.ts +1 -1
- package/templates/src/mcp-server/tools/definitions/echo-app.app-tool.ts +2 -3
- package/templates/src/mcp-server/tools/definitions/echo.tool.ts +1 -1
- package/dist/utils/telemetry/index.d.ts +0 -12
- package/dist/utils/telemetry/index.d.ts.map +0 -1
- package/dist/utils/telemetry/index.js +0 -12
- package/dist/utils/telemetry/index.js.map +0 -1
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: code-simplifier
|
|
3
3
|
description: >
|
|
4
|
-
|
|
4
|
+
Cleanup pass that edits the working tree — over a session's uncommitted changes, or a named path or whole codebase. Reads `git diff` (or the named target) and simplifies, consolidates, and aligns code with the existing codebase — modernize syntax, cut unnecessary complexity and slop, consolidate duplicated logic, catch efficiency issues. Not a bug hunt: defects are reported, not fixed. Use after a substantive working session, or when asked to clean up, simplify, reduce slop, consolidate, modernize, tighten up, de-slop, or scan a codebase. For `@cyanheads/mcp-ts-core` projects, includes specific transformations for tool/resource/prompt definitions, the ctx pattern, error factories, and framework idioms.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "1.
|
|
7
|
+
version: "1.6"
|
|
8
8
|
audience: external
|
|
9
9
|
type: workflow
|
|
10
10
|
---
|
|
@@ -23,7 +23,7 @@ Cleanup pass over a session's changes or a named target. Reviews the code in sco
|
|
|
23
23
|
|
|
24
24
|
Two scopes; the caller's wording picks one, and the diff is the default.
|
|
25
25
|
|
|
26
|
-
- **Diff** (nothing named): run `git status` to see the shape of the working tree, then `git diff HEAD` for all uncommitted changes (staged and unstaged). Untracked files never appear in the diff —
|
|
26
|
+
- **Diff** (nothing named): run `git status` to see the shape of the working tree, then `git diff HEAD` for all uncommitted changes (staged and unstaged). Untracked files never appear in the diff — list them with `git ls-files --others --exclude-standard` (`git status` collapses a new directory to one line) and read them directly. If the diff is empty and there are no untracked files, review the last commit (`git diff HEAD~1 HEAD`); if that is also empty, say the tree is clean and stop. Don't go hunting through the codebase for files to improve.
|
|
27
27
|
- **Target** (a named path, module, or "the whole codebase"): the named files are the scope, whatever their git state. Work one module or directory at a time and re-run the gate after each, so a large scan never becomes one unverifiable diff. Take the target as named — don't rank or narrow it by commit history.
|
|
28
28
|
|
|
29
29
|
### Phase 2: Understand the surrounding codebase
|
|
@@ -33,7 +33,7 @@ Don't review changes in isolation. Before any modifications:
|
|
|
33
33
|
1. **Read the full files** containing changes — not just the diff hunks. Understand imports, surrounding logic, module structure.
|
|
34
34
|
2. **Identify the project language(s)** and select the relevant transformation rules. Discard inapplicable rules.
|
|
35
35
|
3. **Survey adjacent code** — shared utilities, sibling modules, common patterns. You need to know what already exists before deciding something is missing.
|
|
36
|
-
4. **Run the project's gate once before editing** to establish a baseline.
|
|
36
|
+
4. **Run the project's gate once before editing** to establish a baseline. Use the gate the project's `CLAUDE.md` / `AGENTS.md` names; absent one, take it from `package.json` scripts — `devcheck` if present, else `check`, else the separate `typecheck` / `lint` scripts — or a `Makefile` `check` target. Python projects gate on `uv run ruff check`, `uv run ruff format --check`, and the configured type checker. Add the test suite when the gate doesn't run it — read the script rather than assume (a `devcheck` often stops at lint and typecheck); without tests, nothing shows behavior survived the pass. In a Bun project that tests with Vitest, run `bun run test` — bare `bun test` bypasses the script and runs Bun's own runner. If the gate is already red, say so in the summary and don't attribute the failure to your changes.
|
|
37
37
|
|
|
38
38
|
### Phase 3: Review
|
|
39
39
|
|
|
@@ -50,13 +50,15 @@ Evaluate the changes across these dimensions. Not every dimension applies to eve
|
|
|
50
50
|
|
|
51
51
|
- **Redundant state** — State that duplicates existing state, cached values that could be derived.
|
|
52
52
|
- **Unnecessary complexity** — Deep nesting that could be guard clauses, premature abstractions, over-engineered solutions to simple problems.
|
|
53
|
+
- **Speculative generality** — Options, parameters, config flags, generic type parameters, and branches that no caller exercises. Flexibility for a hypothetical caller is cost paid now: remove it, and let the first real use add it back. On a published package's public surface it is API — note it instead (see Dead code).
|
|
53
54
|
- **Pass-through layers** — Apply the deletion test to a wrapper, helper, or module: if deleting it and inlining its body makes the complexity vanish, it was a pass-through — inline it. If the same logic would reappear across several callers, it earns its keep. An interface, port, or injected dependency with a single implementation and no test double is a hypothetical seam, not a real one — collapse it until something actually varies across it.
|
|
54
55
|
- **Test-only reach** — A function extracted or exported only so a test can get at it is a shape problem, not a cleanup: name it in the summary with the module it belongs to. Don't restructure it here — the tests would have to move with it.
|
|
55
|
-
- **Dead code** — Unreachable branches, unused variables, commented-out code. An export nothing imports is dead in an application or a package-internal module; on a published package's public surface it is API — leave it and note it in the summary.
|
|
56
|
+
- **Dead code** — Unreachable branches, unused variables, commented-out code, and debug leftovers from the session (`console.log`, `print`, `debugger`) that aren't the program's real output or the project's logger. An export nothing imports is dead in an application or a package-internal module; on a published package's public surface it is API — leave it and note it in the summary.
|
|
56
57
|
- **Defensive code for impossible states** — Guards for cases the type system or upstream validation already prevents. Drop them.
|
|
57
|
-
- **Type escapes** — `any`, `as` casts that paper over a mismatch, non-null `!`,
|
|
58
|
+
- **Type escapes** — `any`, `as` casts that paper over a mismatch, non-null `!`, `@ts-ignore`, and Python's `# type: ignore` / `cast()`. Each is a claim the compiler couldn't check: replace with a narrowed type, a type guard, or a parse at the boundary. Keep the ones documenting a genuine type-system or third-party-types limitation — confirm the limitation is gone before removing one — and prefer `@ts-expect-error` with a one-line reason over `@ts-ignore`.
|
|
58
59
|
- **Swallowed errors** — Empty `catch {}`, `catch { return null }`, and `try` blocks that log and continue. A fallback that hides a failure is worse than the crash it prevents: rethrow or let it propagate. When wrapping, preserve the chain (`new Error(msg, { cause })`, `raise X from err`).
|
|
59
|
-
- **
|
|
60
|
+
- **Masking defaults** — `?? ''`, `|| []`, `?? 0`, `.get(key, {})` standing in for a value that must exist. The default turns a missing config key or a broken upstream into quietly wrong output further down. When the type already rules out absence, the default is dead — drop it; when the value is optional in the type but required in fact, replace the default with an error that names what's missing, where the value is read. Keep defaults only where absence is a legitimate, expected state.
|
|
61
|
+
- **Comment noise** — Strip comments that restate the code and comments describing behavior the diff removed. Keep file headers, export JSDoc, and any comment carrying a *why* — a constraint, a workaround, an upstream bug reference.
|
|
60
62
|
- **Outdated patterns** — Verbose or legacy syntax where modern equivalents exist. See the transformation tables below.
|
|
61
63
|
|
|
62
64
|
#### Efficiency
|
|
@@ -90,24 +92,31 @@ Evaluate the changes across these dimensions. Not every dimension applies to eve
|
|
|
90
92
|
3. **Correctness bugs are not this pass's job.** A real defect doesn't get folded into a cleanup diff — name it in the summary with file and line so it can be handled as its own change.
|
|
91
93
|
4. **Transform incrementally** — one category of change at a time (modernize syntax, then reduce nesting, then consolidate).
|
|
92
94
|
5. **Verify equivalence** — all functionality, types, and public interfaces must remain unchanged. Re-run the gate from Phase 2 after transforming; a simplification that breaks the build is worse than the verbosity it removed.
|
|
93
|
-
6. **Keep the diff minimal.** Only touch lines that have a real reason to change. Don't reformat untouched code, add comments to code you didn't modify, or "improve" things that are already fine. Formatting belongs to the formatter (Biome, ruff): never hand-adjust whitespace, quotes, or import order
|
|
94
|
-
7. **Never stage, commit, tag, or
|
|
95
|
+
6. **Keep the diff minimal.** Only touch lines that have a real reason to change. Don't reformat untouched code, add comments to code you didn't modify, or "improve" things that are already fine. Formatting belongs to the formatter (Biome, ruff): never hand-adjust whitespace, quotes, or import order. Hunks the project's formatter writes during a gate run stay, even outside the scope — reverting them only fights the next run; mention them in the summary.
|
|
96
|
+
7. **Never stage, commit, tag, push, or stash.** This pass ends with a dirty working tree and a summary; landing the changes is the caller's call. A stash hides the very changes under review — compare against the baseline with `git diff`, never by setting work aside.
|
|
95
97
|
|
|
96
|
-
When done,
|
|
98
|
+
When done, report:
|
|
99
|
+
|
|
100
|
+
- **Gate** — the result before and after the pass, so a failure that predates the cleanup isn't pinned on it.
|
|
101
|
+
- **Fixed** — what changed, grouped by category.
|
|
102
|
+
- **Skipped** — findings deliberately left, each with its reason.
|
|
103
|
+
- **Defects and recommendations** — correctness bugs and out-of-scope changes, each with `file:line`.
|
|
104
|
+
|
|
105
|
+
When nothing earned a change, say the code was already clean.
|
|
97
106
|
|
|
98
107
|
## Common transformations
|
|
99
108
|
|
|
100
109
|
The tables below cover TypeScript and Python. For other languages, apply analogous principles: prefer modern idioms, reduce nesting, eliminate dead code, follow project conventions. Check the project's language floor (`tsconfig` target/lib, `pyproject` `requires-python`) before applying a version-gated row.
|
|
101
110
|
|
|
102
|
-
### TypeScript (modern ESM
|
|
111
|
+
### TypeScript (modern ESM)
|
|
103
112
|
|
|
104
113
|
| Before | After | Why |
|
|
105
114
|
| --- | --- | --- |
|
|
106
115
|
| `const x: Foo = { ... } as Foo` | `const x = { ... } satisfies Foo` | Type-checked without assertion |
|
|
107
|
-
| `let resource = acquire(); try { ... } finally { release(resource) }` | `using resource = acquire()` | Explicit resource
|
|
116
|
+
| `let resource = acquire(); try { ... } finally { release(resource) }` | `using resource = acquire()` | Explicit resource management (TS 5.2+) — only when the resource implements `Symbol.dispose` (`await using` for `Symbol.asyncDispose`); otherwise the `try`/`finally` stays |
|
|
108
117
|
| `if (x !== null && x !== undefined)` | `if (x != null)` | Idiomatic null/undefined check |
|
|
109
118
|
| `arr.filter(x => x !== null) as T[]` | `arr.filter(x => x != null)` | TS 5.5+ infers the type predicate — no cast; on older TS use an explicit `(x): x is T` predicate |
|
|
110
|
-
| `
|
|
119
|
+
| `import { foo } from './index.js'` (a module importing through its own barrel) | `import { foo } from './foo.js'` | Inside a module, import siblings directly — routing through the module's own barrel invites import cycles. Across modules, the public barrel is the right entry point; leave those imports alone |
|
|
111
120
|
| `import { readFile } from 'fs/promises'` | `import { readFile } from 'node:fs/promises'` | `node:` protocol — unambiguous, lint-enforced in Biome |
|
|
112
121
|
| `async function f() { const a = await x(); const b = await y(); }` | `const [a, b] = await Promise.all([x(), y()])` | Parallel when independent |
|
|
113
122
|
| `value \|\| fallback` | `value ?? fallback` | `\|\|` also swallows `0`, `''`, and `false` — use `??` unless every falsy value really should take the fallback |
|
|
@@ -116,10 +125,14 @@ The tables below cover TypeScript and Python. For other languages, apply analogo
|
|
|
116
125
|
| `try { risky() } catch (e: any) { ... }` | `try { risky() } catch (e) { ... }` | Under `strict` the catch binding is already `unknown`; narrow with a type guard before use |
|
|
117
126
|
| `catch (err) { throw new Error('load failed') }` | `throw new Error('load failed', { cause: err })` | Preserve the cause chain |
|
|
118
127
|
| `[...arr].sort(cmp)` / `arr.slice().sort(cmp)` | `arr.toSorted(cmp)` | Non-mutating array methods (ES2023) — also `toReversed`, `toSpliced`, `with` |
|
|
128
|
+
| `arr[arr.length - 1]` | `arr.at(-1)` | ES2022 — typed `T \| undefined`: equivalent under `noUncheckedIndexedAccess`, a new `undefined` to handle otherwise |
|
|
129
|
+
| `arr.reduce((acc, x) => { (acc[key(x)] ??= []).push(x); return acc }, {})` | `Object.groupBy(arr, key)` | ES2024 — returns a null-prototype object whose values are typed `T[] \| undefined`; `Map.groupBy` for non-string keys |
|
|
130
|
+
| `let resolve!: (v: T) => void; const p = new Promise<T>((r) => { resolve = r })` | `const { promise, resolve, reject } = Promise.withResolvers<T>()` | ES2024 — the deferred without the captured-variable dance |
|
|
131
|
+
| `new Set([...a].filter((x) => b.has(x)))` | `a.intersection(b)` | ES2025 `Set` methods — also `union`, `difference`, `symmetricDifference`, `isSubsetOf`; the receiver must be a `Set` |
|
|
119
132
|
| `const c = new AbortController(); setTimeout(() => c.abort(), ms)` | `AbortSignal.timeout(ms)` | Built-in timeout signal; combine with a caller's signal via `AbortSignal.any([...])` |
|
|
120
|
-
| `JSON.parse(JSON.stringify(x))` | `structuredClone(x)` | Deep clone that
|
|
133
|
+
| `JSON.parse(JSON.stringify(x))` | `structuredClone(x)` | Deep clone that keeps Date, Map, Set, and cycles. Not a drop-in: it throws on functions and strips class prototypes, and Dates stay Dates instead of becoming strings |
|
|
121
134
|
| `enum Status { A, B, C }` | `const Status = { A: 'A', B: 'B', C: 'C' } as const` | `enum`, `namespace`, and constructor parameter properties are non-erasable syntax rejected by TS 5.8 `erasableSyntaxOnly` and Node type-stripping — but switching numeric values to strings changes serialized output; keep values stable if they're persisted |
|
|
122
|
-
| `function f(a: string, b: string, c: string, d?: string)` | `function f(opts: FnOptions)` |
|
|
135
|
+
| `function f(a: string, b: string, c: string, d?: string)` | `function f(opts: FnOptions)` | Internal functions whose same-typed positional params can be swapped and still type-check. An exported signature is API — leave it |
|
|
123
136
|
| `throw new Error('Bad input')` (in a tool handler) | `throw validationError('Bad input', { field: 'x' })` | Use framework error factories so the framework can classify and instrument |
|
|
124
137
|
| `const ATTR_KEY = 'mcp.tool.name'` | `import { ATTR_MCP_TOOL_NAME } from '@cyanheads/mcp-ts-core/utils'` | Use framework attribute constants |
|
|
125
138
|
|
|
@@ -134,13 +147,14 @@ The tables below cover TypeScript and Python. For other languages, apply analogo
|
|
|
134
147
|
| `if isinstance(x, Foo): a = x.a; b = x.b` | `match x: case Foo(a=a, b=b): ...` | Structural pattern matching (3.10+) where it destructures — not as a replacement for a flat equality `if/elif` chain |
|
|
135
148
|
| `class Config: def __init__(self, a, b, c): self.a = a ...` | `@dataclass(slots=True) class Config: a: str; b: int; c: float` | Less boilerplate, built-in eq/repr; `frozen=True` when instances shouldn't mutate |
|
|
136
149
|
| `results = []; for item in items: results.append(transform(item))` | `results = [transform(item) for item in items]` | Idiomatic comprehension |
|
|
150
|
+
| `[items[i:i + n] for i in range(0, len(items), n)]` | `itertools.batched(items, n)` | 3.12+ — works on any iterable, not just sequences; yields tuples, not lists |
|
|
137
151
|
| `f = open('x'); try: ... finally: f.close()` | `with open('x') as f: ...` | Context manager for resources |
|
|
138
152
|
| `os.path.join(d, n)`, `os.path.exists(p)`, `open(p).read()` | `Path(d) / n`, `p.exists()`, `p.read_text()` | `pathlib` over `os.path` string juggling |
|
|
139
153
|
| `datetime.utcnow()` / `datetime.utcfromtimestamp(t)` | `datetime.now(UTC)` / `datetime.fromtimestamp(t, UTC)` | Deprecated in 3.12 — the old calls return naive datetimes that compare wrong against aware ones |
|
|
140
|
-
| `zip(a, b)` | `zip(a, b, strict=True)` | 3.10+ — silently truncating
|
|
154
|
+
| `zip(a, b)` where the inputs must match in length | `zip(a, b, strict=True)` | 3.10+ — a mismatch raises instead of silently truncating; plain `zip` stays where truncation is intended |
|
|
141
155
|
| `m = pattern.match(s)` then `if m: use(m)` | `if (m := pattern.match(s)): use(m)` | Walrus operator where it removes a throwaway assignment |
|
|
142
156
|
| `"Hello " + name + "!"` | `f"Hello {name}!"` | f-string over concatenation |
|
|
143
|
-
| `except Exception
|
|
157
|
+
| `except Exception: pass` / `except Exception as e: log(e)` | `except SpecificError:` with real handling, or no `try` at all | Catch only what you can handle; everything else propagates (see Swallowed errors) |
|
|
144
158
|
| `from module import *` | `from module import specific_name` | Explicit imports only |
|
|
145
159
|
| Sequential `await` for independent I/O | `async with asyncio.TaskGroup() as tg: tg.create_task(a()); tg.create_task(b())` | Structured concurrency (3.11+) — cancels siblings on failure and raises an `ExceptionGroup`; `asyncio.gather(..., return_exceptions=True)` stays correct when every result is wanted regardless of failures |
|
|
146
160
|
|
|
@@ -154,7 +168,6 @@ Leave code alone when:
|
|
|
154
168
|
- **Performance-critical paths.** A less readable version may exist for measured performance reasons — check before simplifying.
|
|
155
169
|
- **API compatibility.** Don't change public function signatures, export shapes, or return types that callers depend on.
|
|
156
170
|
- **Tests.** Don't DRY up test code aggressively — test readability and isolation matter more than deduplication.
|
|
157
|
-
- **Type workarounds.** Sometimes an `as` cast or `# type: ignore` exists because of a genuine type system limitation — verify before removing.
|
|
158
171
|
- **The abstraction isn't proven.** Don't create a shared utility for two similar blocks of code. Wait until there are three, and even then only if the abstraction is genuinely simpler than the duplication.
|
|
159
172
|
- **`return await` inside `try` / `finally`.** Collapsing it to `return` is not equivalent — the promise settles outside the block, so `catch` never fires and `finally` runs early. Only strip `await` from a `return` in plain function-body position.
|
|
160
173
|
- **Lazy logging arguments.** `logger.info("loaded %s in %sms", name, ms)` defers formatting until the record is emitted — don't turn it into an f-string.
|
|
@@ -4,7 +4,7 @@ description: >
|
|
|
4
4
|
Design the tool surface, resources, and service layer for a new MCP server. Use when starting a new server, planning a major feature expansion, or when the user describes a domain/API they want to expose via MCP. Produces a design doc at docs/design.md that drives implementation.
|
|
5
5
|
metadata:
|
|
6
6
|
author: cyanheads
|
|
7
|
-
version: "2.
|
|
7
|
+
version: "2.29"
|
|
8
8
|
audience: external
|
|
9
9
|
type: workflow
|
|
10
10
|
---
|
|
@@ -26,6 +26,7 @@ Gather before designing. Ask the user if not obvious from context:
|
|
|
26
26
|
2. **Data sources / source of truth** — APIs, databases, file systems, external services? Or is the server itself the source (in-memory state, pure computation, local-only utility, embedded model)?
|
|
27
27
|
3. **Target users** — what will the LLM (and its human) be trying to accomplish?
|
|
28
28
|
4. **Scope constraints** — read-only? write access? admin operations? what's off-limits?
|
|
29
|
+
5. **Deployment** — local stdio, hosted HTTP, Cloudflare Workers? The answer gates primitives: DataCanvas and `MirrorService` don't run on Workers, and a tool that asks the caller for input mid-call needs a stateful HTTP session (see the Client round-trip row in Step 3).
|
|
29
30
|
|
|
30
31
|
If the domain has a public API, read its docs before designing. For internal-only servers, skip API research and go straight to user goals. Don't design from vibes either way.
|
|
31
32
|
|
|
@@ -63,6 +64,7 @@ Research inline by default — fetch docs, read SDK readmes, confirm assumptions
|
|
|
63
64
|
- Fetch API docs, confirm endpoint availability, auth methods, rate limits
|
|
64
65
|
- Check for official SDKs or client libraries (npm packages)
|
|
65
66
|
- Note any API quirks, pagination patterns, or data format considerations
|
|
67
|
+
- Read the terms of use: whether storing, caching, or redistributing the data is permitted, whether AI or LLM use is, and what attribution the data carries. Note the credential model too — keyless, one operator key, or a key each user supplies. The terms decide whether the server can be hosted for others at all, whether a local mirror is allowed, and what the server instructions must credit.
|
|
66
68
|
|
|
67
69
|
When research is genuinely parallelizable (multiple independent APIs, several SDKs to evaluate), spawn background agents for the independent legs while you proceed with domain mapping. Skip the overhead for a single API — just read it yourself.
|
|
68
70
|
|
|
@@ -75,11 +77,14 @@ When research is genuinely parallelizable (multiple independent APIs, several SD
|
|
|
75
77
|
- **Error shapes** — trigger real 400/404/429 responses to see the actual error format, not just what docs claim.
|
|
76
78
|
- **Unknown-param behavior** — send one deliberately misspelled parameter. If the API silently ignores it (plausible but unfiltered results instead of an error), every typo'd or unverified param name becomes a silent-wrongness bug — the service layer then needs a strict allowlist of confirmed spellings, and new filters require a probe before they ship.
|
|
77
79
|
- **Omission semantics** — for each major optional parameter, check what omitting it actually returns. Some APIs default to the intuitive scope; others silently widen (all historical versions, all statuses, global instead of regional). A default that changes result *meaning* becomes a server-side default plus an echoed output field, not something left to the agent.
|
|
80
|
+
- **Branch frequency, for bundled or bulk data** — when the "API" is a dataset the server ships, probe not only whether each record shape exists but what *fraction* of records takes it: count the records where an optional field is absent, where a value is an array instead of a scalar, where an id maps to many keys rather than one, where a status is missing. A shape found in 0.1% of records is an edge case to handle; one found in 70% is the main path, and a design that specifies only the scalar case leaves the implementer to invent the common one.
|
|
78
81
|
|
|
79
82
|
**Stopping condition:** at minimum, probe one list/search endpoint, one single-item GET, one error case (force a 404 or 400), and one unknown-param request. For large APIs with many resource types, add one probe per major noun. Stop when the response shapes and error envelope are confirmed.
|
|
80
83
|
|
|
81
84
|
This step prevents building a service layer against assumed response shapes that don't match reality.
|
|
82
85
|
|
|
86
|
+
Probe with real data; write the design with made-up data. Every example value that reaches `docs/design.md` — names, emails, phone numbers, hosts, IP addresses, identifiers tied to a person — is synthetic, and no key, token, or private hostname appears in it. The doc is tracked, and git history keeps whatever its first commit carried.
|
|
87
|
+
|
|
83
88
|
### 2. Map User Goals, Then Domain Operations
|
|
84
89
|
|
|
85
90
|
Start with **user goals**, not endpoints. Enumerate the outcomes an agent (and its human) will actually try to accomplish with this server — usually 3–10, scaled to domain size. These drive the workflow tools that form the spine of the surface. Endpoint-inventory-first design produces 1:1 API mirrors; goal-first design produces tools agents reach for. For internal-only servers, goals map to capabilities rather than endpoints — e.g., "format markdown to GFM," "tokenize text by model," "compute file hash."
|
|
@@ -111,7 +116,7 @@ The user-goal list shapes the tool surface; the operation list fills in the gaps
|
|
|
111
116
|
| **App Tool** | **Rare — default to a standard tool.** Only when a human will actively interact with the result in real time *and* the target client supports MCP Apps. Most clients are tool-only and most agent workflows are read-by-LLM, not viewed-by-human. App tools add an iframe + CSP, `app.ontoolresult`/`callServerTool` plumbing, host-context wiring, and a `format()` text twin that still has to be content-complete (since most clients only see that). Two surfaces to keep in sync, two failure modes per change. | Dense tabular state a human scrubs through; form-based human approval in an MCP Apps-capable client |
|
|
112
117
|
| **Resource** | *Additionally* expose as a resource when the data is addressable by stable URI, read-only, and useful as injectable context. | Config, schemas, status, entity-by-ID lookups |
|
|
113
118
|
| **Prompt** | Reusable message template that structures how the LLM approaches a task | Analysis framework, report template, review checklist |
|
|
114
|
-
| **Client round-trip** | Not a registered primitive — a handler *returns* `ctx.requestInput(...)` to ask the client for what only it has, and is re-entered with the answer: a confirmation or form (`inputRequired.elicit`), an authorization or hosted-form URL (`inputRequired.elicitUrl`), the client model's judgment (`inputRequired.createMessage` — borrow the caller's model rather than bundling one), filesystem roots (`inputRequired.listRoots`). Design it into the tool that needs it; see Workflow tool safety and `api-context`. | Destructive-arm confirmation, OAuth consent, summarize-with-the-client's-model |
|
|
119
|
+
| **Client round-trip** | Not a registered primitive — a handler *returns* `ctx.requestInput(...)` to ask the client for what only it has, and is re-entered with the answer: a confirmation or form (`inputRequired.elicit`), an authorization or hosted-form URL (`inputRequired.elicitUrl`), the client model's judgment (`inputRequired.createMessage` — borrow the caller's model rather than bundling one), filesystem roots (`inputRequired.listRoots`). Design it into the tool that needs it; see Workflow tool safety and `api-context`. A server with any such tool declares `createApp({ sessionMode: { default: 'stateful', require: 'stateful' } })` — under stateless HTTP a 2025-era client's round trip is refused, so the tool is unusable rather than guarded. | Destructive-arm confirmation, OAuth consent, summarize-with-the-client's-model |
|
|
115
120
|
| **Neither** | Internal detail, admin-only, not useful to an LLM | Token refresh, webhook setup, migrations |
|
|
116
121
|
|
|
117
122
|
What the tool surface needs to cover depends on the server: a read-only research server has different economics than a CRUD project management server. Consider the domain, the expected agent workflows, whether it wraps one API or many, and what data relationships exist.
|
|
@@ -132,7 +137,7 @@ This is the highest-leverage step. Tool definitions — names, descriptions, par
|
|
|
132
137
|
|
|
133
138
|
#### Tool shapes you'll encounter
|
|
134
139
|
|
|
135
|
-
Most tools follow the `{server}_{verb}_{noun}` default — one focused responsibility, one clear verb, often (but not always) one upstream call. API-wrapping examples: `pubmed_search_articles`, `pubmed_fetch_articles`. Internal-only examples: `markdown_format_text`, `regex_test_pattern`, `tokens_count_text` — same naming convention, no external dep.
|
|
140
|
+
Most tools follow the `{server}_{verb}_{noun}` default — one focused responsibility, one clear verb, often (but not always) one upstream call. API-wrapping examples: `pubmed_search_articles`, `pubmed_fetch_articles`. Internal-only examples: `markdown_format_text`, `regex_test_pattern`, `tokens_count_text` — same naming convention, no external dep. Three variants warrant explicit design pressures of their own:
|
|
136
141
|
|
|
137
142
|
| Shape | Purpose | Typical form | Examples |
|
|
138
143
|
|:------|:--------|:-------------|:---------|
|
|
@@ -168,24 +173,29 @@ Two patterns:
|
|
|
168
173
|
|
|
169
174
|
**Source fallback chains** — try sources in priority order, fall through on failure or empty results. Best when sources cover the *same corpus* with different depth or availability. The output should indicate which source provided the data so the agent (and human) can assess provenance. When the fallback changes what is being searched — a different corpus, different identifiers, different licensing — don't chain: expose the second source as a sibling tool so the agent chooses the corpus knowingly (the shipped `pubmed-mcp-server` keeps `pubmed_europepmc_search` separate for exactly this reason).
|
|
170
175
|
|
|
171
|
-
**Multi-source fan-out** — query multiple sources in parallel, merge results. Best when sources provide complementary data about the same entity. Use `Promise.allSettled` so one
|
|
176
|
+
**Multi-source fan-out** — query multiple sources in parallel, merge results. Best when sources provide complementary data about the same entity. Use `Promise.allSettled` so one *unavailable* source doesn't tank the whole call — but a rejected input is wrong for every source. Rethrow an input-class rejection (`ValidationError`, `InvalidParams`) from any source instead of folding it into a per-source failure: a source that ignores the bad value would otherwise answer for it, and the agent reads a caller mistake as an outage to retry.
|
|
172
177
|
|
|
173
178
|
```ts
|
|
174
179
|
// Handler pseudocode — indicator enrichment across threat intel sources
|
|
175
180
|
async handler(input, ctx) {
|
|
176
|
-
const
|
|
181
|
+
const settled = await Promise.allSettled([
|
|
177
182
|
vtService.lookup(input.indicator),
|
|
178
183
|
abuseIpService.check(input.indicator),
|
|
179
184
|
greynoiseService.query(input.indicator),
|
|
180
185
|
]);
|
|
186
|
+
for (const r of settled) {
|
|
187
|
+
if (r.status === 'rejected' && isInputError(r.reason)) throw r.reason; // the caller's mistake, not an outage
|
|
188
|
+
}
|
|
189
|
+
const [vt, abuse, greynoise] = settled;
|
|
181
190
|
return {
|
|
182
191
|
indicator: input.indicator,
|
|
183
192
|
sources: {
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
193
|
+
// { status: 'ok', data } | { status: 'unavailable', error } — provenance per source
|
|
194
|
+
virustotal: toSourceResult(vt),
|
|
195
|
+
abuseipdb: toSourceResult(abuse),
|
|
196
|
+
greynoise: toSourceResult(greynoise),
|
|
187
197
|
},
|
|
188
|
-
// Server synthesizes a verdict from
|
|
198
|
+
// Server synthesizes a verdict from the sources that answered — the agent gets a conclusion, not raw API dumps
|
|
189
199
|
assessment: synthesizeVerdict(vt, abuse, greynoise),
|
|
190
200
|
};
|
|
191
201
|
}
|
|
@@ -222,12 +232,12 @@ const wrapupInstructions = tool('git_wrapup_instructions', {
|
|
|
222
232
|
output: z.object({
|
|
223
233
|
guidance: z.string()
|
|
224
234
|
.describe('Markdown playbook content, tailored to current account state.'),
|
|
225
|
-
diagnostics: z.record(z.unknown())
|
|
235
|
+
diagnostics: z.record(z.string(), z.unknown())
|
|
226
236
|
.describe('Live state used to tailor the guidance (e.g., staged file count, branch divergence, recent commit cadence).'),
|
|
227
237
|
nextToolSuggestions: z.array(z.object({
|
|
228
238
|
toolName: z.string().describe('Tool to call next.'),
|
|
229
239
|
reason: z.string().describe('Why this step is recommended given current state.'),
|
|
230
|
-
args: z.record(z.unknown()).describe('Arguments pre-filled from diagnostics.'),
|
|
240
|
+
args: z.record(z.string(), z.unknown()).describe('Arguments pre-filled from diagnostics; {} when the tool takes none.'),
|
|
231
241
|
})).describe('Recommended follow-up calls with arguments already populated.'),
|
|
232
242
|
}),
|
|
233
243
|
});
|
|
@@ -235,6 +245,10 @@ const wrapupInstructions = tool('git_wrapup_instructions', {
|
|
|
235
245
|
|
|
236
246
|
Prior art: [`git_wrapup_instructions`](https://github.com/cyanheads/git-mcp-server) walks through staging, commit, and push with repo state inspected. If a server has recurring "how do I do X well given my state" questions, an instruction tool typically beats N topic-specific tools and duplicating guidance in tool descriptions.
|
|
237
247
|
|
|
248
|
+
**One suggestion shape, every server.** Each entry is exactly `{ toolName, reason, args }` — no per-server renames (`tool`, `rationale`, `suggestedArgs`, `arguments`), which force every client to special-case every server. `args` is always present (`{}` for a tool that takes none), and the array is always present, empty when nothing is worth suggesting. A tool on another server is never an entry: this server cannot know it is installed, so name it in the `guidance` prose instead.
|
|
249
|
+
|
|
250
|
+
**Data tools qualify on the same terms.** A data tool carries `nextToolSuggestions` when the right next call depends on its own result and the arguments come from that result — a status sweep that finds a degraded vendor, a connect call that learns which services the upstream supports. A fixed chain (search → get by the returned ID) does not: the IDs in the result and the server instructions already carry it, and a suggestion on every call is token noise.
|
|
251
|
+
|
|
238
252
|
**Suggestions are scoped to what this deployment registers.** A `nextToolSuggestions` entry is an executable call, so it is only correct when its target is enabled under the same configuration — a tool wrapped in `disabledTool()` is absent from `tools/list`, and a suggestion naming it hands the agent a call that fails on dispatch. Build the array from the same config the registration reads, and when the target is off, drop the entry rather than the explanation: `guidance` can still say the capability is unavailable in this deployment and what to do instead. The audit and a worked example live under *Feature-flagged tools* in `add-tool/SKILL.md`.
|
|
239
253
|
|
|
240
254
|
#### Reference tools
|
|
@@ -287,7 +301,8 @@ Descriptions should be as long as needed — concise but complete. Don't artific
|
|
|
287
301
|
Every `.describe()` is prompt text the LLM reads. Parameters should convey: what the value is, what it affects, and (where non-obvious) how to use it well.
|
|
288
302
|
|
|
289
303
|
- **Constrain the type.** Enums and literals over free strings. Regex validation for formatted IDs. Ranges for numeric bounds.
|
|
290
|
-
- **The input root is already strict.** `tool()` applies `.strict()` at the root and advertises `additionalProperties: false`, so an unknown top-level key is rejected by name instead of silently stripped; nested objects still strip unless made strict themselves.
|
|
304
|
+
- **The input root is already strict.** `tool()` applies `.strict()` at the root and advertises `additionalProperties: false`, so an unknown top-level key is rejected by name instead of silently stripped; nested objects still strip unless made strict themselves. Open the root with `.passthrough()` (Zod 4 also spells it `.loose()`; `tool()` honors either) only on a tool that deliberately proxies arbitrary upstream parameters (a raw-query tool), and say so in its description.
|
|
305
|
+
- **A blank optional string is unset.** Form-based clients send every optional field they display as `""`. Treat the blank as omitted — left off the upstream request, never forwarded as `param=`, which some APIs read differently from omission — and never design a `.min(1)` onto an optional field to catch it. `add-tool` has the schema pattern that keeps a validator on the field.
|
|
291
306
|
- **Use JSON-Schema-serializable types only.** The MCP SDK serializes schemas to JSON Schema for `tools/list`. Types like `z.custom()`, `z.date()`, `z.transform()`, `z.bigint()`, `z.symbol()`, `z.void()`, `z.map()`, `z.set()` throw at runtime. Use structural equivalents (e.g., `z.string().describe('ISO 8601 date')` instead of `z.date()`).
|
|
292
307
|
- **Explain costs and tradeoffs** when a parameter choice has meaningful consequences.
|
|
293
308
|
- **Name alternative approaches** when a simpler path exists.
|
|
@@ -331,6 +346,7 @@ The output schema and `format` function control what the LLM reads back. Design
|
|
|
331
346
|
- **Include IDs and references for chaining.** If the agent might act on a result, return the identifiers it needs for follow-up tool calls.
|
|
332
347
|
- **Curate vs. pass-through depends on domain.** Medical/scientific data — don't trim fields that could alter correctness. CRUD responses — return what the agent needs, not the full API payload. Match fidelity to consequence.
|
|
333
348
|
- **Absent upstream data stays absent.** Sparse APIs omit fields; declare those output fields `.optional()` and render "Not available" in `format()` rather than coercing to `false`, `0`, or `""` — a fabricated value is worse than a gap.
|
|
349
|
+
- **Third-party text is data, and `format()` marks it.** When a field carries text other people wrote — posts, reviews, bios, comments, summaries, a user-named place or project — list those fields in the design and plan how `format()` sets them apart: a blockquote or fence for free text, and CR/LF flattened to a space wherever a value is interpolated inline (a heading, a bolded name, a `matched on "…"` line), so a newline in the value can't forge structure in `content[]`. `structuredContent` keeps the value verbatim. Say in the server instructions that this content is data, never instructions; `security-pass` audits the result.
|
|
334
350
|
- **Image and audio bytes ride `ctx.content`, never `output`.** `ctx.content.image(data, mimeType)` / `.audio(...)` emit a `content[]` block once; `output` keeps the metadata the agent reasons over (dimensions, duration, a reference). Base64 in a typed output field ships the bytes twice.
|
|
335
351
|
- **Surface what was done, not just results.** After a write operation, include the post-state so the LLM can chain without an extra round trip.
|
|
336
352
|
- **When the effect lands after the call, wait for it by default.** Some actions are fire-and-forget at the wire — send a wake packet, trigger a job, dispatch a notification, provision a resource — and their immediate result ("sent", "queued", "accepted") answers nothing the agent asked; the agent wants to know whether the machine is up, the job ran, the resource exists. Design the tool to confirm: pre-probe the observable state (cheap; if it is already in the target state, skip the action and say so), act, then poll with early return until the state is observed or a bounded window elapses. Express the window as **one numeric parameter with a default** (`wait_for_s: 30`), where `0` means act and return — not a boolean plus a timeout, which is two parameters for one decision and an awkward default. Every terminal outcome is a *result*, never a throw: `already_<state>`, `<state>` (with elapsed time), `not_<state>` within the window (with `guidance` naming the read-only check tool to re-poll), and `unverified` (window `0`, or nothing to probe). Size the default to cover the common cases while staying inside client tool timeouts, and pair the action with a read-only sibling that probes the same state so an agent can re-check without re-acting.
|
|
@@ -341,7 +357,7 @@ The output schema and `format` function control what the LLM reads back. Design
|
|
|
341
357
|
- **Continuation is a designed field.** Truncation says the cap was hit; continuation says how to get the rest. Return an opaque `cursor` plus `has_more` (via `extractCursor`/`paginateArray` for local sets), and never invent page numbers over a cursor-based upstream — a page the agent can't ask for is a page it will never see.
|
|
342
358
|
- **Spill big *analytical* results to a queryable surface.** When a tool's row set is something an agent would run SQL over *and* can exceed any reasonable context budget — paginated APIs, streamed exports, big query results — pair an inline preview with a `DataCanvas` table holding the full set (`spillover()` in `api-canvas`), and compute distributions or refinement hints across the full result, not the preview, so aggregate signal stays honest. The gates on when a canvas earns its keep are in Step 7.
|
|
343
359
|
- **Outline one large *document* into sections.** When a single tool call returns one document-shaped record (not many rows) that can exceed context — a ~130KB FDA drug label, a big API entity dominated by a few fat fields — return a section *outline* (top-level keys + per-section byte size) instead of truncating, and let the agent re-call with `sections: [...]` to pull only what it needs. `outlineOnOverflow()` (`@cyanheads/mcp-ts-core/utils`) returns a `full | outline` result; pure measure + key-slice, so Cloudflare Workers-portable, unlike canvas-bound `spillover()`. Distinct from spillover on *shape*: spillover splits a row collection, this outlines one fat record. Schema shape and `format()` parity are in the `techniques` skill's `outline-on-overflow` reference.
|
|
344
|
-
- **Mirror a bulk upstream instead of paginating it live.** When the server wraps a large or slow API whose corpus is queried far more than it changes, sync it once into a persistent local index and query that as the primary data path — not the live API per request. Match the backend to corpus size: below ~10⁴ rows → an in-memory index (server-level, no primitive); ~10⁴–10⁷ → the `MirrorService` (embedded SQLite + FTS5; declare a schema + a `sync` ingester via `defineMirror`/`sqliteMirrorStore`, then `runSync`/`query`, see `api-mirror`); above ~10⁷ → an external store. Distinct lifecycle from DataCanvas: a mirror is long-lived and cross-session, refreshed on a schedule; canvas is ephemeral and per-session.
|
|
360
|
+
- **Mirror a bulk upstream instead of paginating it live.** When the server wraps a large or slow API whose corpus is queried far more than it changes, sync it once into a persistent local index and query that as the primary data path — not the live API per request. Two gates come first. The upstream's terms must permit storing the data (Step 1), and a local index must reproduce what the upstream's *search* returns: an API that ranks server-side — field weighting, synonym or ontology expansion — can't be mirrored for search, because the local index answers the same query with different rows and nothing flags it. Exact-key lookups and structured filters mirror faithfully; for rate-limit relief on a ranked-search API, cache responses per request with a short TTL instead. Match the backend to corpus size: below ~10⁴ rows → an in-memory index (server-level, no primitive); ~10⁴–10⁷ → the `MirrorService` (embedded SQLite + FTS5; declare a schema + a `sync` ingester via `defineMirror`/`sqliteMirrorStore`, then `runSync`/`query`, see `api-mirror`); above ~10⁷ → an external store. Distinct lifecycle from DataCanvas: a mirror is long-lived and cross-session, refreshed on a schedule; canvas is ephemeral and per-session.
|
|
345
361
|
- **Two client surfaces, both content-complete.** Different MCP clients forward different surfaces to the model: some (e.g., Claude Code) read `structuredContent` from `output`, others (e.g., Claude Desktop) read `content[]` from `format()`. `format()` is the markdown twin of `structuredContent`, not a summary — a thin `format()` that returns only a count or title leaves `content[]`-only clients blind (the `format-parity` lint catches this). Agent-facing context that is *not* domain payload — empty-result notices, the query as the server parsed it, echoed defaults, totals — goes in the `enrichment` block via `ctx.enrich(...)`, which reaches both surfaces automatically; hand-authored into `format()` text alone it reaches only one. Field-by-field rendering of that block is in the Design table's Enrichment row.
|
|
346
362
|
|
|
347
363
|
#### Batch input design
|
|
@@ -382,7 +398,7 @@ When a tool wraps a complex query language or filter system, provide a simple sh
|
|
|
382
398
|
// text_search handles the common case; query handles everything else
|
|
383
399
|
text_search: z.string().optional()
|
|
384
400
|
.describe('Convenience shortcut: full-text search across title and abstract. For structured filters or field-specific matching, use the query parameter instead.'),
|
|
385
|
-
query: z.record(z.unknown()).optional()
|
|
401
|
+
query: z.record(z.string(), z.unknown()).optional()
|
|
386
402
|
.describe('Full query object for structured filters. Supports operators: _eq, _gt, _and, _or, ...'),
|
|
387
403
|
```
|
|
388
404
|
|
|
@@ -407,18 +423,20 @@ Two params, two behaviors — keep them named distinctly:
|
|
|
407
423
|
|
|
408
424
|
Errors are part of the tool's interface — design them during the design phase, not as an afterthought. Three aspects: **the contract** (which failures are public), **classification** (what error code), and **messaging** (what the LLM reads).
|
|
409
425
|
|
|
410
|
-
**Declare a typed contract for domain failures.** When a tool has known failure modes the agent should plan around (`no_match`, `queue_full`, `vendor_down`), enumerate them as `errors: [{ reason, code, when, recovery, retryable? }]` on the definition. `recovery` is required metadata — the agent's next move when this failure fires (≥ 5 words, lint-validated; spread `ctx.recoveryFor('reason')` into the throw-site `data` to send it on the wire as `data.recovery.hint`). The framework types `ctx.fail(reason, …)` against the declared reason union (typos become TS errors) and auto-populates `data.reason` on the thrown error for stable observability. The error reaches clients with parity across both surfaces — `structuredContent.error` (Claude Code) and `content[]` text (Claude Desktop). Baseline codes (`InternalError`, `ServiceUnavailable`, `Timeout`, `ValidationError`, `SerializationError`) bubble from anywhere and don't need to be enumerated. See `api-errors` skill for the full pattern.
|
|
426
|
+
**Declare a typed contract for domain failures.** When a tool has known failure modes the agent should plan around (`no_match`, `queue_full`, `vendor_down`), enumerate them as `errors: [{ reason, code, when, recovery, retryable? }]` on the definition. `recovery` is required metadata — the agent's next move when this failure fires (≥ 5 words, lint-validated; spread `ctx.recoveryFor('reason')` into the throw-site `data` to send it on the wire as `data.recovery.hint`). The framework types `ctx.fail(reason, …)` against the declared reason union (typos become TS errors) and auto-populates `data.reason` on the thrown error for stable observability. The error reaches clients with parity across both surfaces — `structuredContent.error` (Claude Code) and `content[]` text (Claude Desktop). Baseline codes (`InternalError`, `ServiceUnavailable`, `Timeout`, `ValidationError`, `SerializationError`, `RequestCancelled`) bubble from anywhere and don't need to be enumerated. Mark an entry the service layer throws, rather than the handler, with `thrownBy: 'service'` so the conformance lint doesn't report it as a reason the handler never raises. See `api-errors` skill for the full pattern.
|
|
411
427
|
|
|
412
428
|
**Classify errors by origin.** Different error sources need different codes and different recovery guidance. Map the failure modes for each tool during design:
|
|
413
429
|
|
|
414
430
|
| Origin | Examples | Error code | Agent can recover? |
|
|
415
431
|
|:-------|:---------|:-----------|:-------------------|
|
|
416
432
|
| **Client input** | Bad ID format, invalid params, missing required field, out-of-range value | `ValidationError` | Yes — fix the input and retry |
|
|
417
|
-
| **Upstream API** | 5xx,
|
|
433
|
+
| **Upstream API** | 5xx, network error | `ServiceUnavailable` | Maybe — retry later, or the upstream is down |
|
|
434
|
+
| **Timeout** | Upstream 408/504, a fetch that ran out its timeout, an exhausted retry deadline | `Timeout` | Maybe — retry, or narrow the request so it finishes sooner |
|
|
418
435
|
| **Rate limit** | 429, quota exhausted, queue full | `RateLimited` (`retryable: true`; `withRetry` honors `Retry-After`) | Yes — wait, then retry or reduce frequency |
|
|
419
436
|
| **Not found** | Valid ID format but entity doesn't exist | `NotFound` (or `ValidationError` if ambiguous) | Yes — check the ID, try a search |
|
|
420
437
|
| **Conflict** | Duplicate key, version mismatch, concurrent modification on a write | `Conflict` | Yes — re-read current state, then retry with it |
|
|
421
|
-
| **
|
|
438
|
+
| **Caller auth** | Insufficient scopes, expired token, a rejected key the caller supplied | `Forbidden` / `Unauthorized` | Maybe — escalate or re-auth |
|
|
439
|
+
| **Server credential** | The server's own upstream key is missing, or the upstream rejects it (401/403) | `ConfigurationError` — translated in the service, since the automatic status mapping yields `Unauthorized`/`Forbidden`, which read as the caller's credentials | No — the operator fixes it; the `recovery` names the env var |
|
|
422
440
|
| **Server internal** | Parse failure, missing config, unexpected state | `InternalError` | No — server-side issue |
|
|
423
441
|
|
|
424
442
|
(`InvalidParams` also exists — the framework's `parseToolArguments` emits it when input fails Zod schema validation before the handler runs. Anything the handler itself throws about inputs uses `ValidationError`.)
|
|
@@ -465,15 +483,15 @@ Summarize each tool:
|
|
|
465
483
|
|
|
466
484
|
| Aspect | Decision |
|
|
467
485
|
|:-------|:---------|
|
|
468
|
-
| **Name** | Lowercase snake_case with a canonical server prefix. **3 segments is the strong default** (`{server}_{verb}_{noun}` — e.g., `pubmed_search_articles`, `clinicaltrials_find_eligible`). **2 is fine when the
|
|
486
|
+
| **Name** | Lowercase snake_case with a canonical server prefix. **3 segments is the strong default** (`{server}_{verb}_{noun}` — e.g., `pubmed_search_articles`, `clinicaltrials_find_eligible`). **2 is fine only when the verb is a complete action whose object the domain implies** (`git_pull`, `git_push`, `git_status`, `git_commit` — the remote, working tree, or repo is implicit); don't invent a word to pad those to 3. A verb that takes a caller-specified object — `search`, `find`, `get`, `list`, `query`, `fetch`, `connect`, `create`, `update` — always carries its noun (`ontology_search` → `ontology_search_terms`, `geofeatures_connect` → `geofeatures_connect_endpoint`). Smell test: if `{server}_{verb}` leaves "…what?" unanswered, the noun is missing. **4 is fine when the noun is inherently two words** (`openfda_search_device_clearances`) or the prefix is multi-part. The prefix is judged on clarity, not length: the brand name or the plain well-known word for the domain both pass (`pubmed_`, `patents_`, `earthquake_`); an abbreviation fails only when it reads as something else out of context (`loc_` → lines of code, `ct_` → CT scan). The verb+noun pair should be unambiguous within the server — if two tools could plausibly share a name, the noun isn't specific enough (`read_fulltext` not `read_text` when structured metadata is a separate concept). **Treat name length as a scope smell only when** the extra segment is the *verb* overreaching (e.g., `foo_create_and_send_notification` → split or use modes). |
|
|
469
487
|
| **Granularity** | Scope each tool to one coherent agent action. The implementation can be a single API call (`pubmed_search_articles`), a multi-step workflow, or internal-only — match the unit to the work, don't constrain by call count. |
|
|
470
488
|
| **Description** | Concrete capability statement. Add operational guidance (prerequisites, constraints, gotchas) when non-obvious. |
|
|
471
489
|
| **Input schema** | `.describe()` on every field. Constrained types (enums, literals, regex). Explain costs/tradeoffs of parameter choices. |
|
|
472
490
|
| **Output schema** | Designed for the LLM's next action. Include chaining IDs. Communicate filtering. Post-write state where useful. |
|
|
473
491
|
| **Errors** | Declare domain failure modes as a typed contract (`errors: [{ reason, code, when, recovery, retryable? }]`) so `ctx.fail` is type-checked and capable clients can preview failures via `tools/list`. Every `recovery` string follows the no-dead-ends rule — it names the next tool call. |
|
|
474
|
-
| **Enrichment** | The success-path counterpart to `errors`: declare the agent-facing context fields the handler populates via `ctx.enrich(...)` — zero-hit notice, echoed defaults, totals, truncation — with a kind-tag (`notice`/`total`/`echo`/`delta`) where one fits and an `enrichmentTrailer.render` for any structured field. Keys stay disjoint from `output`. |
|
|
492
|
+
| **Enrichment** | The success-path counterpart to `errors`: declare the agent-facing context fields the handler populates via `ctx.enrich(...)` — zero-hit notice, echoed defaults, totals, truncation — with a kind-tag (`notice`/`total`/`echo`/`delta`) where one fits and an `enrichmentTrailer.render` for any structured field. Keys stay disjoint from `output`. Declare a field required only when every path writes it — a required field one branch skips fails the output parse on every call that takes another branch. |
|
|
475
493
|
| **Annotations** | `readOnlyHint`, `destructiveHint`, `idempotentHint`, `openWorldHint`. Helps clients auto-approve safely. `destructiveHint` defaults to **true** on any tool that isn't read-only, so a benign write must set `destructiveHint: false` explicitly; a read-only tool omits it entirely (`annotation-coherence` lint). |
|
|
476
|
-
| **Auth scopes** | `tool:<snake_tool_name>:<verb>` or `resource:<kebab-resource-name>:<verb>` (e.g., `tool:inventory_search:read`, `resource:echo-app-ui:read`). Domain-led `<domain>:<verb>` (e.g., `inventory:read`) is an acceptable alternative — pick one convention per server and stay consistent. Skip when
|
|
494
|
+
| **Auth scopes** | `tool:<snake_tool_name>:<verb>` or `resource:<kebab-resource-name>:<verb>` (e.g., `tool:inventory_search:read`, `resource:echo-app-ui:read`). Domain-led `<domain>:<verb>` (e.g., `inventory:read`) is an acceptable alternative — pick one convention per server and stay consistent. Skip when no deployment will run `MCP_AUTH_MODE=jwt` or `oauth` — under `none` (stdio, or single-tenant HTTP) scope checks never run. |
|
|
477
495
|
|
|
478
496
|
### 5. Design Resources
|
|
479
497
|
|
|
@@ -513,11 +531,12 @@ For services wrapping external APIs, plan the resilience layer.
|
|
|
513
531
|
|:--------|:---------|
|
|
514
532
|
| **Retry boundary** | Service method wraps full pipeline (fetch + parse), not just the network call. Use `withRetry` from `/utils`. |
|
|
515
533
|
| **Backoff calibration** | Match base delay to upstream recovery time: 200–500ms (ephemeral), 1–2s (rate-limited), 2–5s (degraded). |
|
|
516
|
-
| **HTTP status check** | `fetchWithTimeout` already handles this — non-
|
|
534
|
+
| **HTTP status check** | `fetchWithTimeout` already handles this — a non-2xx throws an `McpError` whose code is mapped from the status (400 → `InvalidParams`, 401 → `Unauthorized`, 403 → `Forbidden`, 404 → `NotFound`, 409 → `Conflict`, 422 → `ValidationError`, 429 → `RateLimited`, 408/504 → `Timeout`, other 5xx → `ServiceUnavailable`; the full table is `api-errors` § *HTTP Response → McpError*), with `status` and `body` on `error.data`, plus `retryAfter` when the upstream sent one. Plan error contracts around those codes, not a blanket `ServiceUnavailable`. |
|
|
517
535
|
| **Parse failure classification** | Response handler detects HTML error pages and throws transient errors, not `SerializationError`. |
|
|
518
536
|
| **Exhausted retry messaging** | `withRetry` enriches the final error with attempt count automatically. |
|
|
519
|
-
| **
|
|
520
|
-
| **
|
|
537
|
+
| **Total deadline** | Retries multiply a per-attempt timeout: four 30s attempts plus backoff outlast a client's 60s request timeout, and the caller gets a transport timeout instead of this server's classified error. Size `withRetry`'s `deadlineMs` to fit inside the client's timeout and thread `attempt.signal` into each fetch (`api-utils`). |
|
|
538
|
+
| **Pacing** | When the upstream mandates a rate (one request per second, N per minute, N concurrent), put a `createPacer` from `/utils` in front of that service — one pacer per upstream budget, composed as `withRetry` outside and the pacer inside — and note the resulting ceiling in the server instructions when it shapes how an agent should batch work. |
|
|
539
|
+
| **Caller-supplied URLs or hosts** | Route through `fetchWithTimeout` with `rejectPrivateIPs: true` — the SSRF guard is off by default. It blocks private, loopback, link-local, and metadata ranges, and is best-effort: DNS rebinding still gets past it, so a deployment that needs hard isolation adds egress controls. Never a bare `fetch` on a caller-controlled destination; `security-pass` audits this sink. |
|
|
521
540
|
|
|
522
541
|
For API efficiency, design the service methods to minimize upstream calls:
|
|
523
542
|
|
|
@@ -559,6 +578,7 @@ What this server does, what system it wraps, who it's for.
|
|
|
559
578
|
|
|
560
579
|
- Bullet list of capabilities and constraints
|
|
561
580
|
- Auth requirements, rate limits, data access scope
|
|
581
|
+
- Deployment targets (stdio / HTTP / Workers), session mode, upstream credential model, and the terms-of-use constraints from Step 1
|
|
562
582
|
|
|
563
583
|
## User Goals
|
|
564
584
|
|
|
@@ -589,13 +609,14 @@ message shape).
|
|
|
589
609
|
|
|
590
610
|
Draft `instructions` string for `createApp()` — the orientation every client sees at
|
|
591
611
|
initialize: canonical workflow chain, identifier semantics, rate-limit posture. A short
|
|
592
|
-
paragraph
|
|
612
|
+
paragraph under 2,048 characters, essentials first — Claude Code truncates the string at that
|
|
613
|
+
length, mid-sentence. Tool descriptions carry the rest.
|
|
593
614
|
|
|
594
615
|
## Implementation Order
|
|
595
616
|
|
|
596
617
|
1. Config and server setup
|
|
597
|
-
2.
|
|
598
|
-
3.
|
|
618
|
+
2. Reference tool (static, no service dependency — grounds field-testing for everything else)
|
|
619
|
+
3. Services (external API clients)
|
|
599
620
|
4. Read-only tools
|
|
600
621
|
5. Write tools
|
|
601
622
|
6. Resources
|
|
@@ -611,7 +632,7 @@ Each step is independently testable.
|
|
|
611
632
|
## API Reference <!-- query language, pagination, rate limits; include when worth documenting -->
|
|
612
633
|
```
|
|
613
634
|
|
|
614
|
-
Keep it concise. The design doc is a working reference, not a spec document — enough to orient a developer (or agent) implementing the server, not more.
|
|
635
|
+
Keep it concise. The design doc is a working reference, not a spec document — enough to orient a developer (or agent) implementing the server, not more. It stays true after the build: when the implementation diverges — a renamed field, a dropped method, a changed error code — the doc changes in the same commit, with the why under Design Decisions. The next agent reads it as the spec.
|
|
615
636
|
|
|
616
637
|
**Workflow Analysis example.** For multi-step workflow tools, document the upstream call sequence in a table — it drives several downstream decisions during implementation: the service-layer method shape, retry boundaries, where cleanup or the confirmation round belongs, and what post-action state to fetch for the response.
|
|
617
638
|
|
|
@@ -644,7 +665,10 @@ Execute the plan using the scaffolding skills:
|
|
|
644
665
|
3. `add-resource` for each standalone resource
|
|
645
666
|
4. `add-prompt` for each prompt
|
|
646
667
|
5. `add-app-tool` *only if any app tools survived the design step* (rare — see the App Tool row in Step 3)
|
|
647
|
-
6. `
|
|
668
|
+
6. `add-test` alongside each definition, declared error contracts included
|
|
669
|
+
7. `devcheck` after each addition
|
|
670
|
+
|
|
671
|
+
Once the surface is built, `tool-defs-analysis` audits the definition language and `field-test` exercises the tools against the live upstream.
|
|
648
672
|
|
|
649
673
|
## Checklist
|
|
650
674
|
|
|
@@ -652,8 +676,8 @@ Items without an `If …:` prefix apply to every design. Conditional items only
|
|
|
652
676
|
|
|
653
677
|
- [ ] Server scope decided — workflow identified, audience sized, boundary drawn (standalone single-API vs. multi-source aggregation vs. internal-only)
|
|
654
678
|
- [ ] **If multi-source:** tool surface organized around user workflows, not API identity. Sources are service-layer details.
|
|
655
|
-
- [ ] External APIs/dependencies researched and verified (docs fetched, SDKs identified)
|
|
656
|
-
- [ ] **If wrapping an external API:** live API probed (at minimum: one list/search, one single-item GET, one error case)
|
|
679
|
+
- [ ] External APIs/dependencies researched and verified (docs fetched, SDKs identified, terms of use read — storage, redistribution, AI use, attribution, credential model)
|
|
680
|
+
- [ ] **If wrapping an external API:** live API probed (at minimum: one list/search, one single-item GET, one error case, one unknown-param request)
|
|
657
681
|
- [ ] User goals enumerated first (3–10 outcomes agents will accomplish, scaled to domain size), then domain operations mapped as raw material
|
|
658
682
|
- [ ] Each operation classified as tool, resource, prompt, or excluded
|
|
659
683
|
- [ ] Catastrophically irreversible operations excluded from the tool surface (stay in vendor UI) — not just `destructiveHint`
|
|
@@ -672,22 +696,25 @@ Items without an `If …:` prefix apply to every design. Conditional items only
|
|
|
672
696
|
- [ ] **If a tool resolves a single identifier:** no-match returns `{ found: false, guidance }` — a result, not a throw — with guidance routing per miss outcome
|
|
673
697
|
- [ ] **If the domain has opaque vocabulary (codes, identifier formats, coverage windows):** reference tool designed (`topic` enum), implemented first, and used as the routing target in recovery strings and notices
|
|
674
698
|
- [ ] Annotations set correctly (`readOnlyHint`, `destructiveHint`, `idempotentHint`, `openWorldHint`) — benign writes set `destructiveHint: false` explicitly, read-only tools omit it
|
|
675
|
-
- [ ] Server-level `instructions` string drafted — workflow chain, identifier semantics, rate-limit posture (ships via `createApp()` on every initialize)
|
|
676
|
-
- [ ] Design doc written to `docs/design.md`
|
|
677
|
-
- [ ] Design confirmed with user (or user pre-authorized implementation)
|
|
699
|
+
- [ ] Server-level `instructions` string drafted — workflow chain, identifier semantics, rate-limit posture, under 2,048 characters (ships via `createApp()` on every initialize)
|
|
678
700
|
- [ ] **If ops share a noun:** related operations consolidated under one tool with a `mode`/`operation` enum — as a `z.discriminatedUnion` input when the arms need different required fields
|
|
679
701
|
- [ ] **If an upstream API has no native search but the relevant set is bounded:** MCP-side list filtering considered — a distinct local filter param (`filter`/`nameContains`, not `query`), filtering the full set, strict token match (fuzzy only when a caller needs typo tolerance)
|
|
680
702
|
- [ ] **If the server has workflow tools:** call-flow documented (upstream sequence + mode arms) in design doc's Workflow Analysis
|
|
681
703
|
- [ ] **If state-aware procedural guidance adds value:** instruction tool considered with `nextToolSuggestions` pre-filled from diagnostics
|
|
682
704
|
- [ ] **If any tool is config-gated:** nothing routes to it while the gate is off — recovery strings, notices, and `guidance` name a callable target or state the capability is unavailable, and structured follow-ups naming it are emitted only under the config that registers it
|
|
683
705
|
- [ ] **If workflow tools have destructive modes:** destructive arm gated on a `ctx.requestInput` confirmation read back from `ctx.inputs`, with `destructiveHint` annotation so clients that never fulfil the round still surface the risk
|
|
706
|
+
- [ ] **If any tool calls `ctx.requestInput`:** `createApp()` declares `sessionMode` with `require: 'stateful'`
|
|
707
|
+
- [ ] **If any output carries text other people wrote:** those fields listed, `format()` quotes or fences free text and flattens CR/LF in inline slots, and the server instructions say the content is data
|
|
684
708
|
- [ ] **If a parameter determines blast radius:** safe default set (e.g., `mode: 'preview'`, `dryRun: true`, `confirmCount` required)
|
|
685
709
|
- [ ] **If an action's effect is observable only after the call (wake, trigger, dispatch, provision):** confirmation on by default through one numeric window param (`0` = act and return), pre-probe then poll with early return, every outcome a result rather than a throw, and a read-only sibling tool that probes the same state
|
|
686
710
|
- [ ] **App tools default to no.** If one was proposed, verified there's a real human-in-the-loop in an MCP Apps-capable client justifying the iframe/CSP/`format()`-twin maintenance cost — otherwise dropped in favor of a standard tool
|
|
687
711
|
- [ ] **If the server exposes resources:** URIs use `{param}` templates, pagination planned for large lists
|
|
688
712
|
- [ ] **If the server is itself the source of truth (no external API):** state lifecycle planned — tenant-scoped vs. global, TTLs, what survives restart, storage backend chosen
|
|
689
713
|
- [ ] **If the server has external deps or shared state:** service layer planned (or explicitly skipped with reasoning)
|
|
690
|
-
- [ ] **If services wrap external APIs:** resilience planned (retry boundary, backoff, parse classification)
|
|
691
|
-
- [ ] **If multi-source server:** each source has its own service with independent auth/retry/rate-limit config. Fallback chains or fan-out strategy documented per tool. Output includes source provenance.
|
|
714
|
+
- [ ] **If services wrap external APIs:** resilience planned (retry boundary, backoff, parse classification, a total deadline inside the client timeout, a pacer where the upstream mandates a rate, `rejectPrivateIPs` on caller-supplied URLs)
|
|
715
|
+
- [ ] **If multi-source server:** each source has its own service with independent auth/retry/rate-limit config. Fallback chains or fan-out strategy documented per tool. Output includes source provenance. An input rejection from any source fails the call rather than reading as that source's outage.
|
|
716
|
+
- [ ] **If mirroring a bulk upstream:** the terms permit storing the data, and a local index reproduces the upstream's search — no server-side ranking or query expansion on the mirrored path
|
|
692
717
|
- [ ] **If exposing a SQL/analytical workspace is in scope:** DataCanvas considered (`api-canvas` skill), and it earns its keep on *analytical* fit (an agent would SQL it), not row count — a discovery/search surface of categorical metadata doesn't qualify. Any tool emitting a `canvas_id` is paired with a `dataframe_query` (+ `dataframe_describe`) tool in the same surface — a token with no query tool is dead output
|
|
693
718
|
- [ ] **If the server needs runtime config:** env vars identified in `server-config.ts`
|
|
719
|
+
- [ ] Design doc written to `docs/design.md` — every example value synthetic, no keys, tokens, or private hosts
|
|
720
|
+
- [ ] Design confirmed with user (or user pre-authorized implementation)
|