mcp-medic 1.2.2 → 1.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +64 -8
- package/dist/checks/index.d.ts +7 -0
- package/dist/checks/index.js +21 -0
- package/dist/checks/malformed-schema.js +17 -0
- package/dist/checks/quality-prompts.d.ts +8 -0
- package/dist/checks/quality-prompts.js +121 -0
- package/dist/checks/quality-resources.d.ts +8 -0
- package/dist/checks/quality-resources.js +92 -0
- package/dist/checks/quality-tool-annotations.d.ts +10 -0
- package/dist/checks/quality-tool-annotations.js +63 -0
- package/dist/checks/quality-tool-descriptions.d.ts +2 -0
- package/dist/checks/quality-tool-descriptions.js +106 -0
- package/dist/checks/quality-tool-names.d.ts +2 -0
- package/dist/checks/quality-tool-names.js +121 -0
- package/dist/checks/quality-tool-output-schema.d.ts +10 -0
- package/dist/checks/quality-tool-output-schema.js +76 -0
- package/dist/checks/quality-tool-surface.d.ts +13 -0
- package/dist/checks/quality-tool-surface.js +99 -0
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +40 -3
- package/dist/diagnostics.d.ts +11 -0
- package/dist/diagnostics.js +28 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.js +2 -0
- package/dist/orchestrator.js +7 -1
- package/dist/policy.d.ts +9 -0
- package/dist/policy.js +72 -0
- package/dist/protocol/connect.js +42 -1
- package/dist/quality-score.d.ts +33 -0
- package/dist/quality-score.js +163 -0
- package/dist/report.d.ts +3 -0
- package/dist/report.js +32 -0
- package/dist/types.d.ts +63 -1
- package/package.json +1 -1
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
/** MCP's `Tool.name` (from `BaseMetadata`) has no documented pattern or
|
|
2
|
+
* length constraint in the spec — so none of these are "protocol
|
|
3
|
+
* violations" in the strict sense. Empty and duplicate names are still
|
|
4
|
+
* treated as errors here because a tool a client cannot reference or
|
|
5
|
+
* distinguish is functionally broken, not merely unstylish; everything
|
|
6
|
+
* else is a warning. */
|
|
7
|
+
const MAX_REASONABLE_NAME_LENGTH = 128;
|
|
8
|
+
/** Conservative, curated list — exact (case-insensitive) matches only, to
|
|
9
|
+
* avoid false-positiving on legitimately short/plain real tool names. */
|
|
10
|
+
const PLACEHOLDER_NAMES = new Set([
|
|
11
|
+
'tool',
|
|
12
|
+
'test',
|
|
13
|
+
'temp',
|
|
14
|
+
'foo',
|
|
15
|
+
'bar',
|
|
16
|
+
'function',
|
|
17
|
+
'func',
|
|
18
|
+
'untitled',
|
|
19
|
+
'new_tool',
|
|
20
|
+
'newtool',
|
|
21
|
+
'example',
|
|
22
|
+
'sample',
|
|
23
|
+
'todo',
|
|
24
|
+
'tbd',
|
|
25
|
+
'xxx',
|
|
26
|
+
]);
|
|
27
|
+
function hasInvalidCharacters(name) {
|
|
28
|
+
// Whitespace (beyond a single space) and control characters break most
|
|
29
|
+
// client UIs and tool-name-as-identifier assumptions, even though the
|
|
30
|
+
// spec doesn't forbid them outright.
|
|
31
|
+
// eslint-disable-next-line no-control-regex
|
|
32
|
+
return /[\t\n\r\x00-\x08\x0b\x0c\x0e-\x1f]/.test(name);
|
|
33
|
+
}
|
|
34
|
+
export const qualityToolNamesCheck = {
|
|
35
|
+
id: 'quality.tool-name',
|
|
36
|
+
description: 'Flags empty, duplicate, overly long, or placeholder-looking tool names.',
|
|
37
|
+
run(connection) {
|
|
38
|
+
const results = [];
|
|
39
|
+
try {
|
|
40
|
+
if (!connection.tools || !Array.isArray(connection.tools)) {
|
|
41
|
+
return results;
|
|
42
|
+
}
|
|
43
|
+
const seen = new Map();
|
|
44
|
+
for (const tool of connection.tools) {
|
|
45
|
+
const name = tool.name;
|
|
46
|
+
seen.set(name, (seen.get(name) ?? 0) + 1);
|
|
47
|
+
if (name.trim() === '') {
|
|
48
|
+
results.push({
|
|
49
|
+
checkId: 'quality.tool-name',
|
|
50
|
+
severity: 'error',
|
|
51
|
+
message: 'Tool has an empty name — a client cannot reference or distinguish it.',
|
|
52
|
+
serverName: connection.server.name,
|
|
53
|
+
toolName: name,
|
|
54
|
+
category: 'quality',
|
|
55
|
+
suggestedFix: { description: 'Give the tool a non-empty, descriptive name.' },
|
|
56
|
+
});
|
|
57
|
+
continue;
|
|
58
|
+
}
|
|
59
|
+
if (name.length > MAX_REASONABLE_NAME_LENGTH) {
|
|
60
|
+
results.push({
|
|
61
|
+
checkId: 'quality.tool-name',
|
|
62
|
+
severity: 'warning',
|
|
63
|
+
message: `Tool "${name.slice(0, 40)}..." name is ${name.length} characters, exceeding the recommended ${MAX_REASONABLE_NAME_LENGTH}.`,
|
|
64
|
+
serverName: connection.server.name,
|
|
65
|
+
toolName: name,
|
|
66
|
+
category: 'quality',
|
|
67
|
+
confidence: 'medium',
|
|
68
|
+
suggestedFix: { description: 'Shorten the tool name to something concise and memorable.' },
|
|
69
|
+
});
|
|
70
|
+
}
|
|
71
|
+
if (hasInvalidCharacters(name)) {
|
|
72
|
+
results.push({
|
|
73
|
+
checkId: 'quality.tool-name',
|
|
74
|
+
severity: 'warning',
|
|
75
|
+
message: `Tool name "${JSON.stringify(name)}" contains whitespace/control characters that may break client tooling.`,
|
|
76
|
+
serverName: connection.server.name,
|
|
77
|
+
toolName: name,
|
|
78
|
+
category: 'quality',
|
|
79
|
+
suggestedFix: { description: 'Use only plain, printable characters in tool names (letters, digits, -, _).' },
|
|
80
|
+
});
|
|
81
|
+
}
|
|
82
|
+
if (name.length === 1 || PLACEHOLDER_NAMES.has(name.trim().toLowerCase())) {
|
|
83
|
+
results.push({
|
|
84
|
+
checkId: 'quality.tool-name',
|
|
85
|
+
severity: 'warning',
|
|
86
|
+
message: `Tool name "${name}" is ambiguous or looks like a placeholder — it doesn't communicate what the tool does.`,
|
|
87
|
+
serverName: connection.server.name,
|
|
88
|
+
toolName: name,
|
|
89
|
+
category: 'quality',
|
|
90
|
+
confidence: 'medium',
|
|
91
|
+
suggestedFix: { description: 'Rename the tool to describe its action, e.g. "search_flights" instead of "tool".' },
|
|
92
|
+
});
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
for (const [name, count] of seen) {
|
|
96
|
+
if (count > 1) {
|
|
97
|
+
results.push({
|
|
98
|
+
checkId: 'quality.tool-name',
|
|
99
|
+
severity: 'error',
|
|
100
|
+
message: `Tool name "${name}" is declared ${count} times — a client cannot reliably invoke a specific one by name.`,
|
|
101
|
+
serverName: connection.server.name,
|
|
102
|
+
toolName: name,
|
|
103
|
+
category: 'quality',
|
|
104
|
+
details: { duplicateCount: count },
|
|
105
|
+
suggestedFix: { description: `Rename duplicate "${name}" tools so every tool name is unique.` },
|
|
106
|
+
});
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
catch (err) {
|
|
111
|
+
results.push({
|
|
112
|
+
checkId: 'quality.tool-name',
|
|
113
|
+
severity: 'error',
|
|
114
|
+
message: `check failed internally: ${err instanceof Error ? err.message : String(err)}`,
|
|
115
|
+
serverName: connection.server.name,
|
|
116
|
+
category: 'quality',
|
|
117
|
+
});
|
|
118
|
+
}
|
|
119
|
+
return results;
|
|
120
|
+
},
|
|
121
|
+
};
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { Check } from '../types.js';
|
|
2
|
+
/**
|
|
3
|
+
* `outputSchema` is optional per the MCP spec (2025-06-18+) — this check
|
|
4
|
+
* never flags its absence. It only inspects an `outputSchema` that IS
|
|
5
|
+
* present, and only for basic structural validity (same shape rules as
|
|
6
|
+
* `inputSchema`: must be an object, should declare "object" as its type).
|
|
7
|
+
* Malformed output schemas are a quality warning, not a protocol error,
|
|
8
|
+
* since a client can simply ignore an outputSchema it can't parse.
|
|
9
|
+
*/
|
|
10
|
+
export declare const qualityToolOutputSchemaCheck: Check;
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `outputSchema` is optional per the MCP spec (2025-06-18+) — this check
|
|
3
|
+
* never flags its absence. It only inspects an `outputSchema` that IS
|
|
4
|
+
* present, and only for basic structural validity (same shape rules as
|
|
5
|
+
* `inputSchema`: must be an object, should declare "object" as its type).
|
|
6
|
+
* Malformed output schemas are a quality warning, not a protocol error,
|
|
7
|
+
* since a client can simply ignore an outputSchema it can't parse.
|
|
8
|
+
*/
|
|
9
|
+
export const qualityToolOutputSchemaCheck = {
|
|
10
|
+
id: 'quality.output-schema',
|
|
11
|
+
description: 'Flags tool outputSchemas that are present but structurally malformed. Never flags a missing outputSchema — it is optional.',
|
|
12
|
+
run(connection) {
|
|
13
|
+
const results = [];
|
|
14
|
+
try {
|
|
15
|
+
if (!connection.tools || !Array.isArray(connection.tools)) {
|
|
16
|
+
return results;
|
|
17
|
+
}
|
|
18
|
+
for (const tool of connection.tools) {
|
|
19
|
+
const schema = tool.outputSchema;
|
|
20
|
+
if (schema === undefined)
|
|
21
|
+
continue; // optional; absence is fine
|
|
22
|
+
if (schema === null || typeof schema !== 'object' || Array.isArray(schema)) {
|
|
23
|
+
const actualType = schema === null ? 'null' : Array.isArray(schema) ? 'array' : typeof schema;
|
|
24
|
+
results.push({
|
|
25
|
+
checkId: 'quality.output-schema',
|
|
26
|
+
severity: 'warning',
|
|
27
|
+
message: `Tool "${tool.name}" declares an outputSchema, but it is ${actualType} instead of a JSON Schema object.`,
|
|
28
|
+
serverName: connection.server.name,
|
|
29
|
+
toolName: tool.name,
|
|
30
|
+
category: 'quality',
|
|
31
|
+
details: { actualType },
|
|
32
|
+
suggestedFix: {
|
|
33
|
+
description: 'Replace outputSchema with a valid JSON Schema object, or remove it if not needed.',
|
|
34
|
+
},
|
|
35
|
+
});
|
|
36
|
+
continue;
|
|
37
|
+
}
|
|
38
|
+
const schemaObj = schema;
|
|
39
|
+
const hasType = typeof schemaObj.type === 'string' || Array.isArray(schemaObj.type);
|
|
40
|
+
const hasCombinatorOrRef = Boolean(schemaObj.$ref) || Boolean(schemaObj.oneOf) || Boolean(schemaObj.anyOf) || Boolean(schemaObj.allOf);
|
|
41
|
+
if (!hasType && !hasCombinatorOrRef) {
|
|
42
|
+
results.push({
|
|
43
|
+
checkId: 'quality.output-schema',
|
|
44
|
+
severity: 'warning',
|
|
45
|
+
message: `Tool "${tool.name}" outputSchema is missing a "type" or combinator field.`,
|
|
46
|
+
serverName: connection.server.name,
|
|
47
|
+
toolName: tool.name,
|
|
48
|
+
category: 'quality',
|
|
49
|
+
suggestedFix: { description: 'Add `"type": "object"` to the outputSchema.' },
|
|
50
|
+
});
|
|
51
|
+
}
|
|
52
|
+
else if (typeof schemaObj.type === 'string' && schemaObj.type !== 'object') {
|
|
53
|
+
results.push({
|
|
54
|
+
checkId: 'quality.output-schema',
|
|
55
|
+
severity: 'warning',
|
|
56
|
+
message: `Tool "${tool.name}" outputSchema declares type "${schemaObj.type}" instead of "object".`,
|
|
57
|
+
serverName: connection.server.name,
|
|
58
|
+
toolName: tool.name,
|
|
59
|
+
category: 'quality',
|
|
60
|
+
suggestedFix: { description: 'Change outputSchema\'s top-level "type" to "object".' },
|
|
61
|
+
});
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
catch (err) {
|
|
66
|
+
results.push({
|
|
67
|
+
checkId: 'quality.output-schema',
|
|
68
|
+
severity: 'error',
|
|
69
|
+
message: `check failed internally: ${err instanceof Error ? err.message : String(err)}`,
|
|
70
|
+
serverName: connection.server.name,
|
|
71
|
+
category: 'quality',
|
|
72
|
+
});
|
|
73
|
+
}
|
|
74
|
+
return results;
|
|
75
|
+
},
|
|
76
|
+
};
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import type { Check } from '../types.js';
|
|
2
|
+
export declare const DEFAULT_MAX_TOOLS_WARNING_THRESHOLD = 100;
|
|
3
|
+
/**
|
|
4
|
+
* Factory rather than a static Check, so the tool-count threshold can be
|
|
5
|
+
* overridden by policy (see `policy.ts`'s `quality.maxTools`) without
|
|
6
|
+
* needing a second, parallel check — same pattern `createPolicyChecks`
|
|
7
|
+
* already uses for configurable org rules.
|
|
8
|
+
*/
|
|
9
|
+
export declare function createToolSurfaceCheck(options?: {
|
|
10
|
+
maxTools?: number;
|
|
11
|
+
}): Check;
|
|
12
|
+
/** Default instance (threshold 100) for the standard `allChecks` list. */
|
|
13
|
+
export declare const qualityToolSurfaceCheck: Check;
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
export const DEFAULT_MAX_TOOLS_WARNING_THRESHOLD = 100;
|
|
2
|
+
function normalizeForSimilarity(name) {
|
|
3
|
+
return name.toLowerCase().replace(/[^a-z0-9]/g, '');
|
|
4
|
+
}
|
|
5
|
+
/**
|
|
6
|
+
* Factory rather than a static Check, so the tool-count threshold can be
|
|
7
|
+
* overridden by policy (see `policy.ts`'s `quality.maxTools`) without
|
|
8
|
+
* needing a second, parallel check — same pattern `createPolicyChecks`
|
|
9
|
+
* already uses for configurable org rules.
|
|
10
|
+
*/
|
|
11
|
+
export function createToolSurfaceCheck(options = {}) {
|
|
12
|
+
const maxTools = options.maxTools ?? DEFAULT_MAX_TOOLS_WARNING_THRESHOLD;
|
|
13
|
+
return {
|
|
14
|
+
id: 'quality.tool-surface',
|
|
15
|
+
description: `Flags excessive tool counts (default warning threshold: ${DEFAULT_MAX_TOOLS_WARNING_THRESHOLD}) and near-duplicate tool names/descriptions.`,
|
|
16
|
+
run(connection) {
|
|
17
|
+
const results = [];
|
|
18
|
+
try {
|
|
19
|
+
const tools = connection.tools;
|
|
20
|
+
if (!tools || !Array.isArray(tools) || tools.length === 0) {
|
|
21
|
+
return results;
|
|
22
|
+
}
|
|
23
|
+
if (tools.length > maxTools) {
|
|
24
|
+
results.push({
|
|
25
|
+
checkId: 'quality.tool-surface',
|
|
26
|
+
severity: 'warning',
|
|
27
|
+
message: `Server exposes ${tools.length} tools, exceeding the ${maxTools}-tool threshold. Large tool surfaces can increase agent/tool-selection complexity.`,
|
|
28
|
+
serverName: connection.server.name,
|
|
29
|
+
category: 'quality',
|
|
30
|
+
details: { toolCount: tools.length, threshold: maxTools },
|
|
31
|
+
suggestedFix: {
|
|
32
|
+
description: 'Consider splitting this server, consolidating overlapping tools, or raising the policy threshold if this is intentional.',
|
|
33
|
+
},
|
|
34
|
+
});
|
|
35
|
+
}
|
|
36
|
+
// Near-duplicate names: same normalized form but not byte-identical.
|
|
37
|
+
// (Exact duplicates are quality.tool-name's job.)
|
|
38
|
+
const byNormalized = new Map();
|
|
39
|
+
for (const tool of tools) {
|
|
40
|
+
const key = normalizeForSimilarity(tool.name);
|
|
41
|
+
if (!key)
|
|
42
|
+
continue;
|
|
43
|
+
const group = byNormalized.get(key) ?? [];
|
|
44
|
+
group.push(tool.name);
|
|
45
|
+
byNormalized.set(key, group);
|
|
46
|
+
}
|
|
47
|
+
for (const group of byNormalized.values()) {
|
|
48
|
+
const distinct = [...new Set(group)];
|
|
49
|
+
if (distinct.length > 1) {
|
|
50
|
+
results.push({
|
|
51
|
+
checkId: 'quality.tool-surface',
|
|
52
|
+
severity: 'warning',
|
|
53
|
+
message: `Tools ${distinct.map((n) => `"${n}"`).join(', ')} have near-identical names — this can confuse tool selection.`,
|
|
54
|
+
serverName: connection.server.name,
|
|
55
|
+
category: 'quality',
|
|
56
|
+
details: { similarNames: distinct },
|
|
57
|
+
suggestedFix: { description: 'Rename these tools to be clearly distinct from one another.' },
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
// Repeated, non-trivial descriptions shared by 3+ tools.
|
|
62
|
+
const byDescription = new Map();
|
|
63
|
+
for (const tool of tools) {
|
|
64
|
+
const desc = tool.description?.trim();
|
|
65
|
+
if (!desc || desc.length < 15)
|
|
66
|
+
continue; // trivial/short strings collide legitimately
|
|
67
|
+
const names = byDescription.get(desc) ?? [];
|
|
68
|
+
names.push(tool.name);
|
|
69
|
+
byDescription.set(desc, names);
|
|
70
|
+
}
|
|
71
|
+
for (const [desc, names] of byDescription) {
|
|
72
|
+
if (names.length >= 3) {
|
|
73
|
+
results.push({
|
|
74
|
+
checkId: 'quality.tool-surface',
|
|
75
|
+
severity: 'warning',
|
|
76
|
+
message: `${names.length} tools (${names.map((n) => `"${n}"`).join(', ')}) share the exact same description — each tool should describe its own specific behavior.`,
|
|
77
|
+
serverName: connection.server.name,
|
|
78
|
+
category: 'quality',
|
|
79
|
+
details: { sharedDescription: desc, toolNames: names },
|
|
80
|
+
suggestedFix: { description: 'Write a distinct, specific description for each of these tools.' },
|
|
81
|
+
});
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
catch (err) {
|
|
86
|
+
results.push({
|
|
87
|
+
checkId: 'quality.tool-surface',
|
|
88
|
+
severity: 'error',
|
|
89
|
+
message: `check failed internally: ${err instanceof Error ? err.message : String(err)}`,
|
|
90
|
+
serverName: connection.server.name,
|
|
91
|
+
category: 'quality',
|
|
92
|
+
});
|
|
93
|
+
}
|
|
94
|
+
return results;
|
|
95
|
+
},
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
/** Default instance (threshold 100) for the standard `allChecks` list. */
|
|
99
|
+
export const qualityToolSurfaceCheck = createToolSurfaceCheck();
|
package/dist/cli.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
export interface ParsedArgs {
|
|
3
|
-
command: 'check' | 'watch' | 'check-all' | 'diff' | 'fix' | 'help' | 'version';
|
|
3
|
+
command: 'check' | 'watch' | 'check-all' | 'diff' | 'fix' | 'score' | 'help' | 'version';
|
|
4
4
|
configPath?: string;
|
|
5
5
|
configPathB?: string;
|
|
6
6
|
globPattern?: string;
|
|
@@ -18,6 +18,8 @@ export interface ParsedArgs {
|
|
|
18
18
|
failOn: 'error' | 'warning';
|
|
19
19
|
checkFilter?: string;
|
|
20
20
|
dryRun: boolean;
|
|
21
|
+
/** Include the MCP quality score section in the report (`--score`, or implied by the `score` command). */
|
|
22
|
+
showScore: boolean;
|
|
21
23
|
/** "auto" (default) or an explicit MCP protocolVersion string, e.g. "2025-06-18". */
|
|
22
24
|
protocolVersion: string;
|
|
23
25
|
}
|
package/dist/cli.js
CHANGED
|
@@ -14,6 +14,7 @@ import { loadPolicy, createPolicyChecks } from './policy.js';
|
|
|
14
14
|
import { runFleetChecks, diffConfigs, filterDiagnosticsByBaseline } from './fleet.js';
|
|
15
15
|
import { formatReportJUnit, formatFleetReportJUnit } from './junit.js';
|
|
16
16
|
import { formatReportSarif } from './sarif.js';
|
|
17
|
+
import { computeReportQualityScore } from './quality-score.js';
|
|
17
18
|
import { SUPPORTED_PROTOCOL_VERSIONS } from './protocol/versions.js';
|
|
18
19
|
import pc from 'picocolors';
|
|
19
20
|
export function parseArgs(argv) {
|
|
@@ -25,6 +26,7 @@ export function parseArgs(argv) {
|
|
|
25
26
|
failOn: 'error',
|
|
26
27
|
dryRun: false,
|
|
27
28
|
protocolVersion: 'auto',
|
|
29
|
+
showScore: false,
|
|
28
30
|
};
|
|
29
31
|
const positional = [];
|
|
30
32
|
for (let i = 0; i < argv.length; i++) {
|
|
@@ -41,6 +43,9 @@ export function parseArgs(argv) {
|
|
|
41
43
|
else if (arg === '--dry-run') {
|
|
42
44
|
args.dryRun = true;
|
|
43
45
|
}
|
|
46
|
+
else if (arg === '--score') {
|
|
47
|
+
args.showScore = true;
|
|
48
|
+
}
|
|
44
49
|
else if (arg === '--help' || arg === '-h') {
|
|
45
50
|
args.command = 'help';
|
|
46
51
|
}
|
|
@@ -114,7 +119,7 @@ export function parseArgs(argv) {
|
|
|
114
119
|
}
|
|
115
120
|
if (args.command !== 'help' && args.command !== 'version') {
|
|
116
121
|
const first = positional[0];
|
|
117
|
-
if (first === 'check' || first === 'watch' || first === 'check-all' || first === 'diff' || first === 'fix') {
|
|
122
|
+
if (first === 'check' || first === 'watch' || first === 'check-all' || first === 'diff' || first === 'fix' || first === 'score') {
|
|
118
123
|
args.command = first;
|
|
119
124
|
if (first === 'diff') {
|
|
120
125
|
args.configPath = positional[1];
|
|
@@ -128,6 +133,9 @@ export function parseArgs(argv) {
|
|
|
128
133
|
args.configPath = positional[1];
|
|
129
134
|
}
|
|
130
135
|
}
|
|
136
|
+
if (first === 'score') {
|
|
137
|
+
args.showScore = true;
|
|
138
|
+
}
|
|
131
139
|
}
|
|
132
140
|
else if (first && !args.configPath) {
|
|
133
141
|
args.configPath = first;
|
|
@@ -183,6 +191,7 @@ USAGE
|
|
|
183
191
|
$ mcp-medic [check] [path/to/config.json] [options]
|
|
184
192
|
$ mcp-medic check --registry <server-id> [options]
|
|
185
193
|
$ mcp-medic check-all "<glob-pattern>" [options]
|
|
194
|
+
$ mcp-medic score <path/to/config.json> [options]
|
|
186
195
|
$ mcp-medic diff <configA.json> <configB.json>
|
|
187
196
|
$ mcp-medic watch <path/to/config.json> [options]
|
|
188
197
|
$ mcp-medic fix <path/to/config.json> [--check <id>] [--dry-run]
|
|
@@ -196,10 +205,11 @@ OPTIONS
|
|
|
196
205
|
--export-junit <file> Export report in JUnit XML format
|
|
197
206
|
--export-json <file> Export report in JSON format
|
|
198
207
|
--export-sarif <file> Export report in SARIF 2.1.0 format (GitHub Code Scanning, etc.)
|
|
208
|
+
--score Include the MCP quality score in the report (implied by the "score" command)
|
|
199
209
|
--show-fixes Print actionable suggested fixes under diagnostics
|
|
200
210
|
--fail-on <severity> Exit with code 1 on 'error' (default) or 'warning'
|
|
201
211
|
--verbose, -v Print raw JSON-RPC traffic and debug messages
|
|
202
|
-
--json Output report in JSON format
|
|
212
|
+
--json Output report in JSON format (always includes a "quality" field)
|
|
203
213
|
--timeout <ms> Per-server handshake timeout in milliseconds (default: 5000)
|
|
204
214
|
--protocol-version <v> MCP protocolVersion to request: "auto" (default, latest supported)
|
|
205
215
|
or an explicit version, e.g. ${SUPPORTED_PROTOCOL_VERSIONS[SUPPORTED_PROTOCOL_VERSIONS.length - 1]}
|
|
@@ -212,6 +222,13 @@ PROTOCOL VERSIONS
|
|
|
212
222
|
older version instead; mcp-medic reports both and fails cleanly if the
|
|
213
223
|
negotiated version isn't one this client supports.
|
|
214
224
|
|
|
225
|
+
MCP QUALITY SCORE
|
|
226
|
+
A deterministic 0-100 score (no LLM, no randomness) across five weighted
|
|
227
|
+
dimensions: Protocol (25%), Schema (20%), Agent usability (20%), Security
|
|
228
|
+
(20%), Reliability (15%). Every point deducted is derived from an actual
|
|
229
|
+
diagnostic and is explained in the report's "Deductions" list. Policy can
|
|
230
|
+
gate on it via "quality": { "minimumScore": N } in .mcp-medic-policy.json.
|
|
231
|
+
|
|
215
232
|
FIX OPTIONS (mcp-medic fix)
|
|
216
233
|
--check <id> Only offer fixes from this check id (e.g. security.untrusted-remote)
|
|
217
234
|
--dry-run Show every available fix as a diff; apply nothing, prompt for nothing
|
|
@@ -266,12 +283,32 @@ async function executeCheck(config, args) {
|
|
|
266
283
|
try {
|
|
267
284
|
const baseline = JSON.parse(readFileSync(args.snapshotPath, 'utf-8'));
|
|
268
285
|
report = filterDiagnosticsByBaseline(report, baseline);
|
|
286
|
+
// The quality score must reflect what's actually being reported —
|
|
287
|
+
// recompute it against the post-baseline-filter diagnostics rather
|
|
288
|
+
// than leaving the pre-filter score (computed by runChecks) stale.
|
|
289
|
+
report.quality = computeReportQualityScore(report);
|
|
269
290
|
}
|
|
270
291
|
catch (err) {
|
|
271
292
|
console.error(pc.yellow(`Warning: Could not read snapshot baseline: ${String(err)}`));
|
|
272
293
|
}
|
|
273
294
|
}
|
|
274
295
|
}
|
|
296
|
+
// Policy gate: fail if the computed quality score is below the configured
|
|
297
|
+
// minimum. This can't be a `Check` (it needs the score computed from every
|
|
298
|
+
// check's output, not a single connection), so it's enforced here instead.
|
|
299
|
+
const policy = loadPolicy(args.policyPath);
|
|
300
|
+
if (typeof policy?.quality?.minimumScore === 'number' && report.quality) {
|
|
301
|
+
if (report.quality.overall < policy.quality.minimumScore) {
|
|
302
|
+
report.diagnostics.push({
|
|
303
|
+
checkId: 'policy.minimum-quality-score',
|
|
304
|
+
severity: 'error',
|
|
305
|
+
message: `MCP quality score ${report.quality.overall} is below the policy minimum of ${policy.quality.minimumScore}.`,
|
|
306
|
+
serverName: config.servers.map((s) => s.name).join(', ') || '(no servers)',
|
|
307
|
+
category: 'configuration',
|
|
308
|
+
});
|
|
309
|
+
report.summary.errors += 1;
|
|
310
|
+
}
|
|
311
|
+
}
|
|
275
312
|
// Handle update snapshot
|
|
276
313
|
if (args.updateSnapshotPath) {
|
|
277
314
|
try {
|
|
@@ -315,7 +352,7 @@ async function executeCheck(config, args) {
|
|
|
315
352
|
console.log(formatReportJSON(report));
|
|
316
353
|
}
|
|
317
354
|
else {
|
|
318
|
-
console.log(colorizeHumanReport(formatReportHuman(report, { showFixes: args.showFixes })));
|
|
355
|
+
console.log(colorizeHumanReport(formatReportHuman(report, { showFixes: args.showFixes, showScore: args.showScore })));
|
|
319
356
|
}
|
|
320
357
|
const hasErrors = report.summary.errors > 0;
|
|
321
358
|
const hasWarnings = report.summary.warnings > 0;
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { DiagnosticCategory, DiagnosticResult } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Infers a stable category from a checkId's namespace prefix, for
|
|
4
|
+
* diagnostics that don't set `category` explicitly. This lets every
|
|
5
|
+
* existing check (schema.*, security.*, policy.*) participate in
|
|
6
|
+
* category-aware reporting/scoring without having to touch each one —
|
|
7
|
+
* only checks written after the taxonomy existed set `category` directly.
|
|
8
|
+
*/
|
|
9
|
+
export declare function inferDiagnosticCategory(checkId: string): DiagnosticCategory;
|
|
10
|
+
/** The effective category of a diagnostic: its own `category` if set, else inferred from `checkId`. */
|
|
11
|
+
export declare function categoryOf(diagnostic: DiagnosticResult): DiagnosticCategory;
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Infers a stable category from a checkId's namespace prefix, for
|
|
3
|
+
* diagnostics that don't set `category` explicitly. This lets every
|
|
4
|
+
* existing check (schema.*, security.*, policy.*) participate in
|
|
5
|
+
* category-aware reporting/scoring without having to touch each one —
|
|
6
|
+
* only checks written after the taxonomy existed set `category` directly.
|
|
7
|
+
*/
|
|
8
|
+
export function inferDiagnosticCategory(checkId) {
|
|
9
|
+
if (checkId.startsWith('protocol.'))
|
|
10
|
+
return 'protocol';
|
|
11
|
+
if (checkId.startsWith('schema.'))
|
|
12
|
+
return 'schema';
|
|
13
|
+
if (checkId.startsWith('security.'))
|
|
14
|
+
return 'security';
|
|
15
|
+
if (checkId.startsWith('usability.'))
|
|
16
|
+
return 'usability';
|
|
17
|
+
if (checkId.startsWith('reliability.'))
|
|
18
|
+
return 'reliability';
|
|
19
|
+
if (checkId.startsWith('quality.'))
|
|
20
|
+
return 'quality';
|
|
21
|
+
if (checkId.startsWith('policy.') || checkId.startsWith('configuration.'))
|
|
22
|
+
return 'configuration';
|
|
23
|
+
return 'quality';
|
|
24
|
+
}
|
|
25
|
+
/** The effective category of a diagnostic: its own `category` if set, else inferred from `checkId`. */
|
|
26
|
+
export function categoryOf(diagnostic) {
|
|
27
|
+
return diagnostic.category ?? inferDiagnosticCategory(diagnostic.checkId);
|
|
28
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -21,4 +21,6 @@ export { allChecks } from './checks/index.js';
|
|
|
21
21
|
export { SUPPORTED_PROTOCOL_VERSIONS, LATEST_SUPPORTED_PROTOCOL_VERSION, KNOWN_UNSUPPORTED_PROTOCOL_VERSIONS, isSupportedProtocolVersion, resolveRequestedProtocolVersion, } from './protocol/versions.js';
|
|
22
22
|
export type { SupportedProtocolVersion, ProtocolVersionNegotiation } from './protocol/versions.js';
|
|
23
23
|
export { isSecretKey, redactRecord, redactDeep, sanitizeServerConfig } from './redact.js';
|
|
24
|
+
export { inferDiagnosticCategory, categoryOf } from './diagnostics.js';
|
|
25
|
+
export { computeConnectionQualityScore, computeReportQualityScore, QUALITY_DIMENSIONS, QUALITY_DIMENSION_WEIGHTS, } from './quality-score.js';
|
|
24
26
|
export * from './types.js';
|
package/dist/index.js
CHANGED
|
@@ -13,4 +13,6 @@ export { formatReportSarif } from './sarif.js';
|
|
|
13
13
|
export { allChecks } from './checks/index.js';
|
|
14
14
|
export { SUPPORTED_PROTOCOL_VERSIONS, LATEST_SUPPORTED_PROTOCOL_VERSION, KNOWN_UNSUPPORTED_PROTOCOL_VERSIONS, isSupportedProtocolVersion, resolveRequestedProtocolVersion, } from './protocol/versions.js';
|
|
15
15
|
export { isSecretKey, redactRecord, redactDeep, sanitizeServerConfig } from './redact.js';
|
|
16
|
+
export { inferDiagnosticCategory, categoryOf } from './diagnostics.js';
|
|
17
|
+
export { computeConnectionQualityScore, computeReportQualityScore, QUALITY_DIMENSIONS, QUALITY_DIMENSION_WEIGHTS, } from './quality-score.js';
|
|
16
18
|
export * from './types.js';
|
package/dist/orchestrator.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { sanitizeServerConfig } from './redact.js';
|
|
2
|
+
import { computeReportQualityScore } from './quality-score.js';
|
|
2
3
|
async function connectStub(config) {
|
|
3
4
|
return {
|
|
4
5
|
server: config,
|
|
@@ -57,10 +58,15 @@ export async function runChecks(config, options = {}) {
|
|
|
57
58
|
errors: diagnostics.filter((d) => d.severity === 'error').length,
|
|
58
59
|
warnings: diagnostics.filter((d) => d.severity === 'warning').length,
|
|
59
60
|
};
|
|
60
|
-
|
|
61
|
+
const report = {
|
|
61
62
|
configSource: config.sourcePath,
|
|
62
63
|
connections,
|
|
63
64
|
diagnostics,
|
|
64
65
|
summary,
|
|
65
66
|
};
|
|
67
|
+
const quality = computeReportQualityScore(report);
|
|
68
|
+
if (quality) {
|
|
69
|
+
report.quality = quality;
|
|
70
|
+
}
|
|
71
|
+
return report;
|
|
66
72
|
}
|
package/dist/policy.d.ts
CHANGED
|
@@ -3,7 +3,16 @@ export interface MCPMedicPolicy {
|
|
|
3
3
|
bannedTransports?: TransportType[];
|
|
4
4
|
allowedDomains?: string[];
|
|
5
5
|
minDescriptionLength?: number;
|
|
6
|
+
/** @deprecated use `quality.requireToolDescriptions` — kept for backward compatibility, same effect. */
|
|
6
7
|
requireToolDescriptions?: boolean;
|
|
8
|
+
quality?: {
|
|
9
|
+
/** Gate: `check`/`check-all` fail (an error diagnostic is added) if the computed MCP quality score falls below this. */
|
|
10
|
+
minimumScore?: number;
|
|
11
|
+
/** Overrides quality.tool-surface's default 100-tool warning threshold with an org-enforced error threshold. */
|
|
12
|
+
maxTools?: number;
|
|
13
|
+
/** Same effect as the top-level (deprecated) `requireToolDescriptions`; this nested form takes precedence if both are set. */
|
|
14
|
+
requireToolDescriptions?: boolean;
|
|
15
|
+
};
|
|
7
16
|
}
|
|
8
17
|
export type MCPDoctorPolicy = MCPMedicPolicy;
|
|
9
18
|
/**
|
package/dist/policy.js
CHANGED
|
@@ -145,5 +145,77 @@ export function createPolicyChecks(policy) {
|
|
|
145
145
|
},
|
|
146
146
|
});
|
|
147
147
|
}
|
|
148
|
+
// 4. Require tool descriptions (nested quality.requireToolDescriptions takes
|
|
149
|
+
// precedence over the deprecated top-level field; either enables this).
|
|
150
|
+
const requireToolDescriptions = policy.quality?.requireToolDescriptions ?? policy.requireToolDescriptions;
|
|
151
|
+
if (requireToolDescriptions) {
|
|
152
|
+
checks.push({
|
|
153
|
+
id: 'policy.require-tool-descriptions',
|
|
154
|
+
description: 'Organizational policy: every tool must have a non-empty description.',
|
|
155
|
+
run(connection) {
|
|
156
|
+
const results = [];
|
|
157
|
+
try {
|
|
158
|
+
if (!connection.tools)
|
|
159
|
+
return results;
|
|
160
|
+
for (const tool of connection.tools) {
|
|
161
|
+
if (!tool.description || tool.description.trim() === '') {
|
|
162
|
+
results.push({
|
|
163
|
+
checkId: 'policy.require-tool-descriptions',
|
|
164
|
+
severity: 'error',
|
|
165
|
+
message: `Tool "${tool.name}" has no description — organizational policy requires one.`,
|
|
166
|
+
serverName: connection.server.name,
|
|
167
|
+
toolName: tool.name,
|
|
168
|
+
category: 'configuration',
|
|
169
|
+
suggestedFix: { description: `Add a description to tool "${tool.name}".` },
|
|
170
|
+
});
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
catch (err) {
|
|
175
|
+
results.push({
|
|
176
|
+
checkId: 'policy.require-tool-descriptions',
|
|
177
|
+
severity: 'error',
|
|
178
|
+
message: `policy check failed internally: ${err instanceof Error ? err.message : String(err)}`,
|
|
179
|
+
serverName: connection.server.name,
|
|
180
|
+
});
|
|
181
|
+
}
|
|
182
|
+
return results;
|
|
183
|
+
},
|
|
184
|
+
});
|
|
185
|
+
}
|
|
186
|
+
// 5. Max tools (org-enforced hard limit; distinct from quality.tool-surface's
|
|
187
|
+
// default 100-tool *warning*, which stays a recommendation, not a policy gate).
|
|
188
|
+
if (typeof policy.quality?.maxTools === 'number' && policy.quality.maxTools > 0) {
|
|
189
|
+
const maxTools = policy.quality.maxTools;
|
|
190
|
+
checks.push({
|
|
191
|
+
id: 'policy.max-tools',
|
|
192
|
+
description: `Organizational policy: a server may expose at most ${maxTools} tools.`,
|
|
193
|
+
run(connection) {
|
|
194
|
+
const results = [];
|
|
195
|
+
try {
|
|
196
|
+
const count = connection.tools?.length ?? 0;
|
|
197
|
+
if (count > maxTools) {
|
|
198
|
+
results.push({
|
|
199
|
+
checkId: 'policy.max-tools',
|
|
200
|
+
severity: 'error',
|
|
201
|
+
message: `Server exposes ${count} tools, exceeding organizational policy limit of ${maxTools}.`,
|
|
202
|
+
serverName: connection.server.name,
|
|
203
|
+
category: 'configuration',
|
|
204
|
+
details: { toolCount: count, maxTools },
|
|
205
|
+
});
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
catch (err) {
|
|
209
|
+
results.push({
|
|
210
|
+
checkId: 'policy.max-tools',
|
|
211
|
+
severity: 'error',
|
|
212
|
+
message: `policy check failed internally: ${err instanceof Error ? err.message : String(err)}`,
|
|
213
|
+
serverName: connection.server.name,
|
|
214
|
+
});
|
|
215
|
+
}
|
|
216
|
+
return results;
|
|
217
|
+
},
|
|
218
|
+
});
|
|
219
|
+
}
|
|
148
220
|
return checks;
|
|
149
221
|
}
|