mcp-medic 1.2.2 → 1.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +82 -8
- package/dist/checks/index.d.ts +8 -0
- package/dist/checks/index.js +24 -0
- package/dist/checks/malformed-schema.js +17 -0
- package/dist/checks/protocol-connection-health.d.ts +17 -0
- package/dist/checks/protocol-connection-health.js +78 -0
- package/dist/checks/quality-prompts.d.ts +8 -0
- package/dist/checks/quality-prompts.js +121 -0
- package/dist/checks/quality-resources.d.ts +8 -0
- package/dist/checks/quality-resources.js +92 -0
- package/dist/checks/quality-tool-annotations.d.ts +10 -0
- package/dist/checks/quality-tool-annotations.js +63 -0
- package/dist/checks/quality-tool-descriptions.d.ts +2 -0
- package/dist/checks/quality-tool-descriptions.js +106 -0
- package/dist/checks/quality-tool-names.d.ts +35 -0
- package/dist/checks/quality-tool-names.js +160 -0
- package/dist/checks/quality-tool-output-schema.d.ts +10 -0
- package/dist/checks/quality-tool-output-schema.js +76 -0
- package/dist/checks/quality-tool-surface.d.ts +13 -0
- package/dist/checks/quality-tool-surface.js +99 -0
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +35 -3
- package/dist/diagnostics.d.ts +11 -0
- package/dist/diagnostics.js +28 -0
- package/dist/index.d.ts +7 -0
- package/dist/index.js +5 -0
- package/dist/orchestrator.js +7 -1
- package/dist/policy.d.ts +9 -0
- package/dist/policy.js +72 -0
- package/dist/protocol/connect.js +42 -1
- package/dist/protocol/quality-rules.d.ts +46 -0
- package/dist/protocol/quality-rules.js +37 -0
- package/dist/quality-score.d.ts +76 -0
- package/dist/quality-score.js +242 -0
- package/dist/report.d.ts +3 -0
- package/dist/report.js +54 -0
- package/dist/types.d.ts +81 -1
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -14,6 +14,7 @@ import { loadPolicy, createPolicyChecks } from './policy.js';
|
|
|
14
14
|
import { runFleetChecks, diffConfigs, filterDiagnosticsByBaseline } from './fleet.js';
|
|
15
15
|
import { formatReportJUnit, formatFleetReportJUnit } from './junit.js';
|
|
16
16
|
import { formatReportSarif } from './sarif.js';
|
|
17
|
+
import { computeReportQualityScore, checkMinimumScorePolicy } from './quality-score.js';
|
|
17
18
|
import { SUPPORTED_PROTOCOL_VERSIONS } from './protocol/versions.js';
|
|
18
19
|
import pc from 'picocolors';
|
|
19
20
|
export function parseArgs(argv) {
|
|
@@ -25,6 +26,7 @@ export function parseArgs(argv) {
|
|
|
25
26
|
failOn: 'error',
|
|
26
27
|
dryRun: false,
|
|
27
28
|
protocolVersion: 'auto',
|
|
29
|
+
showScore: false,
|
|
28
30
|
};
|
|
29
31
|
const positional = [];
|
|
30
32
|
for (let i = 0; i < argv.length; i++) {
|
|
@@ -41,6 +43,9 @@ export function parseArgs(argv) {
|
|
|
41
43
|
else if (arg === '--dry-run') {
|
|
42
44
|
args.dryRun = true;
|
|
43
45
|
}
|
|
46
|
+
else if (arg === '--score') {
|
|
47
|
+
args.showScore = true;
|
|
48
|
+
}
|
|
44
49
|
else if (arg === '--help' || arg === '-h') {
|
|
45
50
|
args.command = 'help';
|
|
46
51
|
}
|
|
@@ -114,7 +119,7 @@ export function parseArgs(argv) {
|
|
|
114
119
|
}
|
|
115
120
|
if (args.command !== 'help' && args.command !== 'version') {
|
|
116
121
|
const first = positional[0];
|
|
117
|
-
if (first === 'check' || first === 'watch' || first === 'check-all' || first === 'diff' || first === 'fix') {
|
|
122
|
+
if (first === 'check' || first === 'watch' || first === 'check-all' || first === 'diff' || first === 'fix' || first === 'score') {
|
|
118
123
|
args.command = first;
|
|
119
124
|
if (first === 'diff') {
|
|
120
125
|
args.configPath = positional[1];
|
|
@@ -128,6 +133,9 @@ export function parseArgs(argv) {
|
|
|
128
133
|
args.configPath = positional[1];
|
|
129
134
|
}
|
|
130
135
|
}
|
|
136
|
+
if (first === 'score') {
|
|
137
|
+
args.showScore = true;
|
|
138
|
+
}
|
|
131
139
|
}
|
|
132
140
|
else if (first && !args.configPath) {
|
|
133
141
|
args.configPath = first;
|
|
@@ -183,6 +191,7 @@ USAGE
|
|
|
183
191
|
$ mcp-medic [check] [path/to/config.json] [options]
|
|
184
192
|
$ mcp-medic check --registry <server-id> [options]
|
|
185
193
|
$ mcp-medic check-all "<glob-pattern>" [options]
|
|
194
|
+
$ mcp-medic score <path/to/config.json> [options]
|
|
186
195
|
$ mcp-medic diff <configA.json> <configB.json>
|
|
187
196
|
$ mcp-medic watch <path/to/config.json> [options]
|
|
188
197
|
$ mcp-medic fix <path/to/config.json> [--check <id>] [--dry-run]
|
|
@@ -196,10 +205,11 @@ OPTIONS
|
|
|
196
205
|
--export-junit <file> Export report in JUnit XML format
|
|
197
206
|
--export-json <file> Export report in JSON format
|
|
198
207
|
--export-sarif <file> Export report in SARIF 2.1.0 format (GitHub Code Scanning, etc.)
|
|
208
|
+
--score Include the MCP quality score in the report (implied by the "score" command)
|
|
199
209
|
--show-fixes Print actionable suggested fixes under diagnostics
|
|
200
210
|
--fail-on <severity> Exit with code 1 on 'error' (default) or 'warning'
|
|
201
211
|
--verbose, -v Print raw JSON-RPC traffic and debug messages
|
|
202
|
-
--json Output report in JSON format
|
|
212
|
+
--json Output report in JSON format (always includes a "quality" field)
|
|
203
213
|
--timeout <ms> Per-server handshake timeout in milliseconds (default: 5000)
|
|
204
214
|
--protocol-version <v> MCP protocolVersion to request: "auto" (default, latest supported)
|
|
205
215
|
or an explicit version, e.g. ${SUPPORTED_PROTOCOL_VERSIONS[SUPPORTED_PROTOCOL_VERSIONS.length - 1]}
|
|
@@ -212,6 +222,13 @@ PROTOCOL VERSIONS
|
|
|
212
222
|
older version instead; mcp-medic reports both and fails cleanly if the
|
|
213
223
|
negotiated version isn't one this client supports.
|
|
214
224
|
|
|
225
|
+
MCP QUALITY SCORE
|
|
226
|
+
A deterministic 0-100 score (no LLM, no randomness) across five weighted
|
|
227
|
+
dimensions: Protocol (25%), Schema (20%), Agent usability (20%), Security
|
|
228
|
+
(20%), Reliability (15%). Every point deducted is derived from an actual
|
|
229
|
+
diagnostic and is explained in the report's "Deductions" list. Policy can
|
|
230
|
+
gate on it via "quality": { "minimumScore": N } in .mcp-medic-policy.json.
|
|
231
|
+
|
|
215
232
|
FIX OPTIONS (mcp-medic fix)
|
|
216
233
|
--check <id> Only offer fixes from this check id (e.g. security.untrusted-remote)
|
|
217
234
|
--dry-run Show every available fix as a diff; apply nothing, prompt for nothing
|
|
@@ -266,12 +283,27 @@ async function executeCheck(config, args) {
|
|
|
266
283
|
try {
|
|
267
284
|
const baseline = JSON.parse(readFileSync(args.snapshotPath, 'utf-8'));
|
|
268
285
|
report = filterDiagnosticsByBaseline(report, baseline);
|
|
286
|
+
// The quality score must reflect what's actually being reported —
|
|
287
|
+
// recompute it against the post-baseline-filter diagnostics rather
|
|
288
|
+
// than leaving the pre-filter score (computed by runChecks) stale.
|
|
289
|
+
report.quality = computeReportQualityScore(report, checks);
|
|
269
290
|
}
|
|
270
291
|
catch (err) {
|
|
271
292
|
console.error(pc.yellow(`Warning: Could not read snapshot baseline: ${String(err)}`));
|
|
272
293
|
}
|
|
273
294
|
}
|
|
274
295
|
}
|
|
296
|
+
// Policy gate: fail if the computed quality score is below the configured
|
|
297
|
+
// minimum. This can't be a `Check` (it needs the score computed from every
|
|
298
|
+
// check's output, not a single connection), so it's enforced here instead.
|
|
299
|
+
const policy = loadPolicy(args.policyPath);
|
|
300
|
+
if (typeof policy?.quality?.minimumScore === 'number' && report.quality) {
|
|
301
|
+
const allServerNames = config.servers.map((s) => s.name).join(', ') || '(no servers)';
|
|
302
|
+
const policyDiagnostics = checkMinimumScorePolicy(report.quality, policy.quality.minimumScore, allServerNames);
|
|
303
|
+
report.diagnostics.push(...policyDiagnostics);
|
|
304
|
+
report.summary.errors += policyDiagnostics.filter((d) => d.severity === 'error').length;
|
|
305
|
+
report.summary.warnings += policyDiagnostics.filter((d) => d.severity === 'warning').length;
|
|
306
|
+
}
|
|
275
307
|
// Handle update snapshot
|
|
276
308
|
if (args.updateSnapshotPath) {
|
|
277
309
|
try {
|
|
@@ -315,7 +347,7 @@ async function executeCheck(config, args) {
|
|
|
315
347
|
console.log(formatReportJSON(report));
|
|
316
348
|
}
|
|
317
349
|
else {
|
|
318
|
-
console.log(colorizeHumanReport(formatReportHuman(report, { showFixes: args.showFixes })));
|
|
350
|
+
console.log(colorizeHumanReport(formatReportHuman(report, { showFixes: args.showFixes, showScore: args.showScore })));
|
|
319
351
|
}
|
|
320
352
|
const hasErrors = report.summary.errors > 0;
|
|
321
353
|
const hasWarnings = report.summary.warnings > 0;
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { DiagnosticCategory, DiagnosticResult } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Infers a stable category from a checkId's namespace prefix, for
|
|
4
|
+
* diagnostics that don't set `category` explicitly. This lets every
|
|
5
|
+
* existing check (schema.*, security.*, policy.*) participate in
|
|
6
|
+
* category-aware reporting/scoring without having to touch each one —
|
|
7
|
+
* only checks written after the taxonomy existed set `category` directly.
|
|
8
|
+
*/
|
|
9
|
+
export declare function inferDiagnosticCategory(checkId: string): DiagnosticCategory;
|
|
10
|
+
/** The effective category of a diagnostic: its own `category` if set, else inferred from `checkId`. */
|
|
11
|
+
export declare function categoryOf(diagnostic: DiagnosticResult): DiagnosticCategory;
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Infers a stable category from a checkId's namespace prefix, for
|
|
3
|
+
* diagnostics that don't set `category` explicitly. This lets every
|
|
4
|
+
* existing check (schema.*, security.*, policy.*) participate in
|
|
5
|
+
* category-aware reporting/scoring without having to touch each one —
|
|
6
|
+
* only checks written after the taxonomy existed set `category` directly.
|
|
7
|
+
*/
|
|
8
|
+
export function inferDiagnosticCategory(checkId) {
|
|
9
|
+
if (checkId.startsWith('protocol.'))
|
|
10
|
+
return 'protocol';
|
|
11
|
+
if (checkId.startsWith('schema.'))
|
|
12
|
+
return 'schema';
|
|
13
|
+
if (checkId.startsWith('security.'))
|
|
14
|
+
return 'security';
|
|
15
|
+
if (checkId.startsWith('usability.'))
|
|
16
|
+
return 'usability';
|
|
17
|
+
if (checkId.startsWith('reliability.'))
|
|
18
|
+
return 'reliability';
|
|
19
|
+
if (checkId.startsWith('quality.'))
|
|
20
|
+
return 'quality';
|
|
21
|
+
if (checkId.startsWith('policy.') || checkId.startsWith('configuration.'))
|
|
22
|
+
return 'configuration';
|
|
23
|
+
return 'quality';
|
|
24
|
+
}
|
|
25
|
+
/** The effective category of a diagnostic: its own `category` if set, else inferred from `checkId`. */
|
|
26
|
+
export function categoryOf(diagnostic) {
|
|
27
|
+
return diagnostic.category ?? inferDiagnosticCategory(diagnostic.checkId);
|
|
28
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -21,4 +21,11 @@ export { allChecks } from './checks/index.js';
|
|
|
21
21
|
export { SUPPORTED_PROTOCOL_VERSIONS, LATEST_SUPPORTED_PROTOCOL_VERSION, KNOWN_UNSUPPORTED_PROTOCOL_VERSIONS, isSupportedProtocolVersion, resolveRequestedProtocolVersion, } from './protocol/versions.js';
|
|
22
22
|
export type { SupportedProtocolVersion, ProtocolVersionNegotiation } from './protocol/versions.js';
|
|
23
23
|
export { isSecretKey, redactRecord, redactDeep, sanitizeServerConfig } from './redact.js';
|
|
24
|
+
export { inferDiagnosticCategory, categoryOf } from './diagnostics.js';
|
|
25
|
+
export { computeConnectionQualityScore, computeReportQualityScore, computeQualityCoverage, checkMinimumScorePolicy, QUALITY_DIMENSIONS, QUALITY_DIMENSION_WEIGHTS, QUALITY_SCORE_DISCLAIMER, } from './quality-score.js';
|
|
26
|
+
export { getProtocolQualityRules } from './protocol/quality-rules.js';
|
|
27
|
+
export type { ProtocolQualityRules, ToolNameProtocolRules } from './protocol/quality-rules.js';
|
|
28
|
+
export { protocolConnectionHealthCheck } from './checks/protocol-connection-health.js';
|
|
29
|
+
export { evaluateToolName } from './checks/quality-tool-names.js';
|
|
30
|
+
export type { ToolNameFinding } from './checks/quality-tool-names.js';
|
|
24
31
|
export * from './types.js';
|
package/dist/index.js
CHANGED
|
@@ -13,4 +13,9 @@ export { formatReportSarif } from './sarif.js';
|
|
|
13
13
|
export { allChecks } from './checks/index.js';
|
|
14
14
|
export { SUPPORTED_PROTOCOL_VERSIONS, LATEST_SUPPORTED_PROTOCOL_VERSION, KNOWN_UNSUPPORTED_PROTOCOL_VERSIONS, isSupportedProtocolVersion, resolveRequestedProtocolVersion, } from './protocol/versions.js';
|
|
15
15
|
export { isSecretKey, redactRecord, redactDeep, sanitizeServerConfig } from './redact.js';
|
|
16
|
+
export { inferDiagnosticCategory, categoryOf } from './diagnostics.js';
|
|
17
|
+
export { computeConnectionQualityScore, computeReportQualityScore, computeQualityCoverage, checkMinimumScorePolicy, QUALITY_DIMENSIONS, QUALITY_DIMENSION_WEIGHTS, QUALITY_SCORE_DISCLAIMER, } from './quality-score.js';
|
|
18
|
+
export { getProtocolQualityRules } from './protocol/quality-rules.js';
|
|
19
|
+
export { protocolConnectionHealthCheck } from './checks/protocol-connection-health.js';
|
|
20
|
+
export { evaluateToolName } from './checks/quality-tool-names.js';
|
|
16
21
|
export * from './types.js';
|
package/dist/orchestrator.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { sanitizeServerConfig } from './redact.js';
|
|
2
|
+
import { computeReportQualityScore } from './quality-score.js';
|
|
2
3
|
async function connectStub(config) {
|
|
3
4
|
return {
|
|
4
5
|
server: config,
|
|
@@ -57,10 +58,15 @@ export async function runChecks(config, options = {}) {
|
|
|
57
58
|
errors: diagnostics.filter((d) => d.severity === 'error').length,
|
|
58
59
|
warnings: diagnostics.filter((d) => d.severity === 'warning').length,
|
|
59
60
|
};
|
|
60
|
-
|
|
61
|
+
const report = {
|
|
61
62
|
configSource: config.sourcePath,
|
|
62
63
|
connections,
|
|
63
64
|
diagnostics,
|
|
64
65
|
summary,
|
|
65
66
|
};
|
|
67
|
+
const quality = computeReportQualityScore(report, checks);
|
|
68
|
+
if (quality) {
|
|
69
|
+
report.quality = quality;
|
|
70
|
+
}
|
|
71
|
+
return report;
|
|
66
72
|
}
|
package/dist/policy.d.ts
CHANGED
|
@@ -3,7 +3,16 @@ export interface MCPMedicPolicy {
|
|
|
3
3
|
bannedTransports?: TransportType[];
|
|
4
4
|
allowedDomains?: string[];
|
|
5
5
|
minDescriptionLength?: number;
|
|
6
|
+
/** @deprecated use `quality.requireToolDescriptions` — kept for backward compatibility, same effect. */
|
|
6
7
|
requireToolDescriptions?: boolean;
|
|
8
|
+
quality?: {
|
|
9
|
+
/** Gate: `check`/`check-all` fail (an error diagnostic is added) if the computed MCP quality score falls below this. */
|
|
10
|
+
minimumScore?: number;
|
|
11
|
+
/** Overrides quality.tool-surface's default 100-tool warning threshold with an org-enforced error threshold. */
|
|
12
|
+
maxTools?: number;
|
|
13
|
+
/** Same effect as the top-level (deprecated) `requireToolDescriptions`; this nested form takes precedence if both are set. */
|
|
14
|
+
requireToolDescriptions?: boolean;
|
|
15
|
+
};
|
|
7
16
|
}
|
|
8
17
|
export type MCPDoctorPolicy = MCPMedicPolicy;
|
|
9
18
|
/**
|
package/dist/policy.js
CHANGED
|
@@ -145,5 +145,77 @@ export function createPolicyChecks(policy) {
|
|
|
145
145
|
},
|
|
146
146
|
});
|
|
147
147
|
}
|
|
148
|
+
// 4. Require tool descriptions (nested quality.requireToolDescriptions takes
|
|
149
|
+
// precedence over the deprecated top-level field; either enables this).
|
|
150
|
+
const requireToolDescriptions = policy.quality?.requireToolDescriptions ?? policy.requireToolDescriptions;
|
|
151
|
+
if (requireToolDescriptions) {
|
|
152
|
+
checks.push({
|
|
153
|
+
id: 'policy.require-tool-descriptions',
|
|
154
|
+
description: 'Organizational policy: every tool must have a non-empty description.',
|
|
155
|
+
run(connection) {
|
|
156
|
+
const results = [];
|
|
157
|
+
try {
|
|
158
|
+
if (!connection.tools)
|
|
159
|
+
return results;
|
|
160
|
+
for (const tool of connection.tools) {
|
|
161
|
+
if (!tool.description || tool.description.trim() === '') {
|
|
162
|
+
results.push({
|
|
163
|
+
checkId: 'policy.require-tool-descriptions',
|
|
164
|
+
severity: 'error',
|
|
165
|
+
message: `Tool "${tool.name}" has no description — organizational policy requires one.`,
|
|
166
|
+
serverName: connection.server.name,
|
|
167
|
+
toolName: tool.name,
|
|
168
|
+
category: 'configuration',
|
|
169
|
+
suggestedFix: { description: `Add a description to tool "${tool.name}".` },
|
|
170
|
+
});
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
catch (err) {
|
|
175
|
+
results.push({
|
|
176
|
+
checkId: 'policy.require-tool-descriptions',
|
|
177
|
+
severity: 'error',
|
|
178
|
+
message: `policy check failed internally: ${err instanceof Error ? err.message : String(err)}`,
|
|
179
|
+
serverName: connection.server.name,
|
|
180
|
+
});
|
|
181
|
+
}
|
|
182
|
+
return results;
|
|
183
|
+
},
|
|
184
|
+
});
|
|
185
|
+
}
|
|
186
|
+
// 5. Max tools (org-enforced hard limit; distinct from quality.tool-surface's
|
|
187
|
+
// default 100-tool *warning*, which stays a recommendation, not a policy gate).
|
|
188
|
+
if (typeof policy.quality?.maxTools === 'number' && policy.quality.maxTools > 0) {
|
|
189
|
+
const maxTools = policy.quality.maxTools;
|
|
190
|
+
checks.push({
|
|
191
|
+
id: 'policy.max-tools',
|
|
192
|
+
description: `Organizational policy: a server may expose at most ${maxTools} tools.`,
|
|
193
|
+
run(connection) {
|
|
194
|
+
const results = [];
|
|
195
|
+
try {
|
|
196
|
+
const count = connection.tools?.length ?? 0;
|
|
197
|
+
if (count > maxTools) {
|
|
198
|
+
results.push({
|
|
199
|
+
checkId: 'policy.max-tools',
|
|
200
|
+
severity: 'error',
|
|
201
|
+
message: `Server exposes ${count} tools, exceeding organizational policy limit of ${maxTools}.`,
|
|
202
|
+
serverName: connection.server.name,
|
|
203
|
+
category: 'configuration',
|
|
204
|
+
details: { toolCount: count, maxTools },
|
|
205
|
+
});
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
catch (err) {
|
|
209
|
+
results.push({
|
|
210
|
+
checkId: 'policy.max-tools',
|
|
211
|
+
severity: 'error',
|
|
212
|
+
message: `policy check failed internally: ${err instanceof Error ? err.message : String(err)}`,
|
|
213
|
+
serverName: connection.server.name,
|
|
214
|
+
});
|
|
215
|
+
}
|
|
216
|
+
return results;
|
|
217
|
+
},
|
|
218
|
+
});
|
|
219
|
+
}
|
|
148
220
|
return checks;
|
|
149
221
|
}
|
package/dist/protocol/connect.js
CHANGED
|
@@ -67,6 +67,23 @@ function validateResponse(response, expectedId) {
|
|
|
67
67
|
}
|
|
68
68
|
return response.result;
|
|
69
69
|
}
|
|
70
|
+
function normalizeToolAnnotations(value) {
|
|
71
|
+
if (!value || typeof value !== 'object' || Array.isArray(value))
|
|
72
|
+
return undefined;
|
|
73
|
+
const record = value;
|
|
74
|
+
const annotations = {};
|
|
75
|
+
if (typeof record.title === 'string')
|
|
76
|
+
annotations.title = record.title;
|
|
77
|
+
if (typeof record.readOnlyHint === 'boolean')
|
|
78
|
+
annotations.readOnlyHint = record.readOnlyHint;
|
|
79
|
+
if (typeof record.destructiveHint === 'boolean')
|
|
80
|
+
annotations.destructiveHint = record.destructiveHint;
|
|
81
|
+
if (typeof record.idempotentHint === 'boolean')
|
|
82
|
+
annotations.idempotentHint = record.idempotentHint;
|
|
83
|
+
if (typeof record.openWorldHint === 'boolean')
|
|
84
|
+
annotations.openWorldHint = record.openWorldHint;
|
|
85
|
+
return annotations;
|
|
86
|
+
}
|
|
70
87
|
function normalizeTools(result) {
|
|
71
88
|
if (!Array.isArray(result.tools))
|
|
72
89
|
throw new Error('tools/list response has no tools array');
|
|
@@ -77,8 +94,11 @@ function normalizeTools(result) {
|
|
|
77
94
|
const value = tool;
|
|
78
95
|
return {
|
|
79
96
|
name: value.name,
|
|
97
|
+
...(typeof value.title === 'string' ? { title: value.title } : {}),
|
|
80
98
|
...(typeof value.description === 'string' ? { description: value.description } : {}),
|
|
81
99
|
inputSchema: value.inputSchema,
|
|
100
|
+
...(value.outputSchema !== undefined ? { outputSchema: value.outputSchema } : {}),
|
|
101
|
+
...(normalizeToolAnnotations(value.annotations) ? { annotations: normalizeToolAnnotations(value.annotations) } : {}),
|
|
82
102
|
};
|
|
83
103
|
});
|
|
84
104
|
}
|
|
@@ -93,11 +113,29 @@ function normalizeResources(result) {
|
|
|
93
113
|
return {
|
|
94
114
|
uri: value.uri,
|
|
95
115
|
...(typeof value.name === 'string' ? { name: value.name } : {}),
|
|
116
|
+
...(typeof value.title === 'string' ? { title: value.title } : {}),
|
|
96
117
|
...(typeof value.description === 'string' ? { description: value.description } : {}),
|
|
97
118
|
...(typeof value.mimeType === 'string' ? { mimeType: value.mimeType } : {}),
|
|
119
|
+
...(typeof value.size === 'number' ? { size: value.size } : {}),
|
|
98
120
|
};
|
|
99
121
|
});
|
|
100
122
|
}
|
|
123
|
+
function normalizePromptArgument(value) {
|
|
124
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) {
|
|
125
|
+
return {};
|
|
126
|
+
}
|
|
127
|
+
const record = value;
|
|
128
|
+
const arg = {};
|
|
129
|
+
if (typeof record.name === 'string')
|
|
130
|
+
arg.name = record.name;
|
|
131
|
+
if (typeof record.title === 'string')
|
|
132
|
+
arg.title = record.title;
|
|
133
|
+
if (typeof record.description === 'string')
|
|
134
|
+
arg.description = record.description;
|
|
135
|
+
if (typeof record.required === 'boolean')
|
|
136
|
+
arg.required = record.required;
|
|
137
|
+
return arg;
|
|
138
|
+
}
|
|
101
139
|
function normalizePrompts(result) {
|
|
102
140
|
if (!Array.isArray(result.prompts))
|
|
103
141
|
throw new Error('prompts/list response has no prompts array');
|
|
@@ -108,8 +146,11 @@ function normalizePrompts(result) {
|
|
|
108
146
|
const value = prompt;
|
|
109
147
|
return {
|
|
110
148
|
name: value.name,
|
|
149
|
+
...(typeof value.title === 'string' ? { title: value.title } : {}),
|
|
111
150
|
...(typeof value.description === 'string' ? { description: value.description } : {}),
|
|
112
|
-
...(Array.isArray(value.arguments)
|
|
151
|
+
...(Array.isArray(value.arguments)
|
|
152
|
+
? { arguments: value.arguments.map(normalizePromptArgument) }
|
|
153
|
+
: {}),
|
|
113
154
|
};
|
|
114
155
|
});
|
|
115
156
|
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Protocol-version-aware quality rules.
|
|
3
|
+
*
|
|
4
|
+
* Some things a quality check might flag are genuine protocol violations
|
|
5
|
+
* for a given negotiated MCP version (hard requirement/prohibition); others
|
|
6
|
+
* are ecosystem recommendations that happen to be entirely valid per spec.
|
|
7
|
+
* Mixing these up means a check can wrongly call a style preference a
|
|
8
|
+
* "protocol violation," or (worse) miss a real one because the version
|
|
9
|
+
* that defines it was never consulted.
|
|
10
|
+
*
|
|
11
|
+
* This module is the single place that answers "does this negotiated
|
|
12
|
+
* version define a hard constraint here?" so quality checks don't each
|
|
13
|
+
* have to know MCP spec history — and so a future protocol version that
|
|
14
|
+
* *does* add a constraint only needs a new entry here, not a rewrite of
|
|
15
|
+
* every check that uses it.
|
|
16
|
+
*
|
|
17
|
+
* Current state of research (see src/protocol/versions.ts for the same
|
|
18
|
+
* source): every version in SUPPORTED_PROTOCOL_VERSIONS (2024-11-05
|
|
19
|
+
* through 2025-11-25) defines `Tool.name` as a plain `string` with no
|
|
20
|
+
* documented length or character-pattern constraint. So today, every
|
|
21
|
+
* version returns the same (empty) rule set — this module exists for the
|
|
22
|
+
* abstraction, not because any current version actually differs from
|
|
23
|
+
* another.
|
|
24
|
+
*/
|
|
25
|
+
export interface ToolNameProtocolRules {
|
|
26
|
+
/** A hard maximum length the negotiated version's spec actually defines
|
|
27
|
+
* for `Tool.name`, if any. Exceeding it is a protocol violation, not a
|
|
28
|
+
* style recommendation. `undefined` = the spec defines no such limit for
|
|
29
|
+
* this version — length-based feedback should be a quality warning, not
|
|
30
|
+
* an error. */
|
|
31
|
+
maxLength?: number;
|
|
32
|
+
/** A hard character-pattern the negotiated version's spec actually
|
|
33
|
+
* requires `Tool.name` to match, if any. `undefined` = no such
|
|
34
|
+
* constraint — character-based feedback should be a quality warning. */
|
|
35
|
+
pattern?: RegExp;
|
|
36
|
+
}
|
|
37
|
+
export interface ProtocolQualityRules {
|
|
38
|
+
toolName: ToolNameProtocolRules;
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Returns the protocol-defined (hard) quality rules for a negotiated MCP
|
|
42
|
+
* version. `version` is normally `MCPConnection.protocolVersion.negotiated`.
|
|
43
|
+
* An unknown or absent version gets the same "no hard constraints" rules
|
|
44
|
+
* as every currently-supported version, rather than guessing.
|
|
45
|
+
*/
|
|
46
|
+
export declare function getProtocolQualityRules(version: string | undefined): ProtocolQualityRules;
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Protocol-version-aware quality rules.
|
|
3
|
+
*
|
|
4
|
+
* Some things a quality check might flag are genuine protocol violations
|
|
5
|
+
* for a given negotiated MCP version (hard requirement/prohibition); others
|
|
6
|
+
* are ecosystem recommendations that happen to be entirely valid per spec.
|
|
7
|
+
* Mixing these up means a check can wrongly call a style preference a
|
|
8
|
+
* "protocol violation," or (worse) miss a real one because the version
|
|
9
|
+
* that defines it was never consulted.
|
|
10
|
+
*
|
|
11
|
+
* This module is the single place that answers "does this negotiated
|
|
12
|
+
* version define a hard constraint here?" so quality checks don't each
|
|
13
|
+
* have to know MCP spec history — and so a future protocol version that
|
|
14
|
+
* *does* add a constraint only needs a new entry here, not a rewrite of
|
|
15
|
+
* every check that uses it.
|
|
16
|
+
*
|
|
17
|
+
* Current state of research (see src/protocol/versions.ts for the same
|
|
18
|
+
* source): every version in SUPPORTED_PROTOCOL_VERSIONS (2024-11-05
|
|
19
|
+
* through 2025-11-25) defines `Tool.name` as a plain `string` with no
|
|
20
|
+
* documented length or character-pattern constraint. So today, every
|
|
21
|
+
* version returns the same (empty) rule set — this module exists for the
|
|
22
|
+
* abstraction, not because any current version actually differs from
|
|
23
|
+
* another.
|
|
24
|
+
*/
|
|
25
|
+
/** No currently-supported MCP version defines a hard length/pattern
|
|
26
|
+
* constraint on Tool.name — see the module doc comment. */
|
|
27
|
+
const NO_HARD_CONSTRAINTS = { toolName: {} };
|
|
28
|
+
/**
|
|
29
|
+
* Returns the protocol-defined (hard) quality rules for a negotiated MCP
|
|
30
|
+
* version. `version` is normally `MCPConnection.protocolVersion.negotiated`.
|
|
31
|
+
* An unknown or absent version gets the same "no hard constraints" rules
|
|
32
|
+
* as every currently-supported version, rather than guessing.
|
|
33
|
+
*/
|
|
34
|
+
export function getProtocolQualityRules(version) {
|
|
35
|
+
void version; // reserved for when a supported version actually defines a constraint
|
|
36
|
+
return NO_HARD_CONSTRAINTS;
|
|
37
|
+
}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
import type { MCPConnection, DiagnosticResult, RunReport, QualityDimension, QualityDeduction, QualityScoreBreakdown, ReportQualityScore, QualityCoverage, Check } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Deterministic MCP quality score. No LLM, no randomness, no network
|
|
4
|
+
* calls beyond the MCP inspection `connect()` already performed — the same
|
|
5
|
+
* report always produces the same score.
|
|
6
|
+
*
|
|
7
|
+
* Design: connection -> diagnostics -> normalized per-checkId deductions ->
|
|
8
|
+
* capped per-dimension score -> weighted overall score. Every deduction is
|
|
9
|
+
* derived from a visible `DiagnosticResult` — nothing is deducted directly
|
|
10
|
+
* from connection metadata that isn't also represented as a diagnostic
|
|
11
|
+
* (see .agent-room/DECISIONS.md, "v1.1 trust hardening" entry, for the
|
|
12
|
+
* `protocol.connection-health` check this replaced).
|
|
13
|
+
*
|
|
14
|
+
* (Result types — QualityDimension, QualityDeduction, QualityScoreBreakdown,
|
|
15
|
+
* ReportQualityScore, QualityCoverage, CoverageStatus — live in src/types.ts
|
|
16
|
+
* alongside RunReport, since RunReport.quality is one of them; this module
|
|
17
|
+
* just implements the math.)
|
|
18
|
+
*/
|
|
19
|
+
export type { QualityDimension, QualityDeduction, QualityScoreBreakdown, ReportQualityScore };
|
|
20
|
+
export declare const QUALITY_DIMENSIONS: readonly QualityDimension[];
|
|
21
|
+
/** Suggested weights from the product spec; documented here since every
|
|
22
|
+
* point deduction must be explainable in terms of these. Do not change
|
|
23
|
+
* without a documented reason — see .agent-room/DECISIONS.md. */
|
|
24
|
+
export declare const QUALITY_DIMENSION_WEIGHTS: Record<QualityDimension, number>;
|
|
25
|
+
/**
|
|
26
|
+
* The score is a measurement, not a certification: it reports what
|
|
27
|
+
* mcp-medic's passive checks actually found (or, per `coverage`, actually
|
|
28
|
+
* looked for) — never a formal audit of the server's real security or
|
|
29
|
+
* behavior. Surfaced in both JSON (`ReportQualityScore.disclaimer`) and the
|
|
30
|
+
* human report.
|
|
31
|
+
*/
|
|
32
|
+
export declare const QUALITY_SCORE_DISCLAIMER: string;
|
|
33
|
+
/**
|
|
34
|
+
* Computes a quality score for one connection, purely from `diagnostics`
|
|
35
|
+
* (filtered to that server) — no connection metadata is consulted
|
|
36
|
+
* directly. Returns `undefined` for a connection that never connected —
|
|
37
|
+
* there's nothing meaningful to score (no tools/schema/capabilities were
|
|
38
|
+
* ever observed); see `computeReportQualityScore` for how the report-level
|
|
39
|
+
* score surfaces that instead of silently dropping it.
|
|
40
|
+
*/
|
|
41
|
+
export declare function computeConnectionQualityScore(connection: MCPConnection, diagnostics: DiagnosticResult[]): QualityScoreBreakdown | undefined;
|
|
42
|
+
/**
|
|
43
|
+
* Coverage answers "was this dimension actually evaluated," independent of
|
|
44
|
+
* what score it got — a 100 from zero checks run means "nothing was
|
|
45
|
+
* looked for," not "nothing is wrong." Derived from the *ids* of the
|
|
46
|
+
* checks that actually ran (`executedChecks`), compared against the
|
|
47
|
+
* built-in `allChecks` reference set grouped by dimension. Reliability is
|
|
48
|
+
* special-cased to always 'covered': its signal (did the connection
|
|
49
|
+
* succeed) isn't produced by any optional `Check` — it's inherent to
|
|
50
|
+
* attempting a connection at all, so it's never "not covered."
|
|
51
|
+
*/
|
|
52
|
+
export declare function computeQualityCoverage(executedChecks: Check[]): QualityCoverage;
|
|
53
|
+
/**
|
|
54
|
+
* Aggregate score across every connected server in a report (simple mean —
|
|
55
|
+
* deliberately not weighted by tool count, so one huge server can't drown
|
|
56
|
+
* out the rest of a fleet). `executedChecks` should be the exact list
|
|
57
|
+
* passed to `runChecks({ checks })` for this report, so `coverage` reflects
|
|
58
|
+
* what actually ran (defaults to the full built-in set if omitted, for
|
|
59
|
+
* callers that haven't been updated to pass it through).
|
|
60
|
+
* Returns `undefined` if no server connected — nothing to score at all.
|
|
61
|
+
*/
|
|
62
|
+
export declare function computeReportQualityScore(report: RunReport, executedChecks?: Check[]): ReportQualityScore | undefined;
|
|
63
|
+
/**
|
|
64
|
+
* Enforces `.mcp-medic-policy.json`'s `quality.minimumScore` as a CI gate.
|
|
65
|
+
* Pure function so it's directly unit-testable without needing a real CLI
|
|
66
|
+
* invocation or a restricted check set — see PHASE 12 ("policy/CI") in
|
|
67
|
+
* .agent-room/DECISIONS.md.
|
|
68
|
+
*
|
|
69
|
+
* Returns diagnostics to append to the report (0, 1, or 2):
|
|
70
|
+
* - an 'error' if the score is below `minimumScore` (fails the run).
|
|
71
|
+
* - a 'warning' if the score was computed from a partial assessment
|
|
72
|
+
* (incomplete coverage, or a server that couldn't be scored) — a
|
|
73
|
+
* minimumScore gate must never silently "pass" as if a full check had
|
|
74
|
+
* run when it didn't.
|
|
75
|
+
*/
|
|
76
|
+
export declare function checkMinimumScorePolicy(quality: ReportQualityScore, minimumScore: number, serverNamesLabel: string): DiagnosticResult[];
|