ratemyagent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,82 @@
1
+ """RateMyAgent: production reliability testing for AI agents, MCP servers, and LLM tools.
2
+
3
+ from ratemyagent import MockTarget, scan
4
+
5
+ result = await scan(MockTarget.healthy())
6
+ print(result.score, result.passed)
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from .models import (
12
+ CheckResult,
13
+ ErrorKind,
14
+ FaultKind,
15
+ Invocation,
16
+ ProbeResult,
17
+ Request,
18
+ Response,
19
+ ScanResult,
20
+ TargetInfo,
21
+ ToolInfo,
22
+ Trajectory,
23
+ )
24
+ from .policy import Policy, PolicyError
25
+ from .probes import (
26
+ BehaviorAnalyzer,
27
+ ConcurrencyTester,
28
+ ContractTester,
29
+ CostAnalyzer,
30
+ FaultInjector,
31
+ Probe,
32
+ ProbeConfig,
33
+ available_probes,
34
+ get_probe,
35
+ )
36
+ from .scanner import PHASES, scan
37
+ from .targets import (
38
+ FaultConfig,
39
+ FaultProxy,
40
+ LLMTarget,
41
+ MCPTarget,
42
+ MockTarget,
43
+ Target,
44
+ build_target,
45
+ )
46
+
47
+ __version__ = "0.1.0"
48
+
49
+ __all__ = [
50
+ "BehaviorAnalyzer",
51
+ "CheckResult",
52
+ "ConcurrencyTester",
53
+ "ContractTester",
54
+ "CostAnalyzer",
55
+ "ErrorKind",
56
+ "FaultConfig",
57
+ "FaultInjector",
58
+ "FaultKind",
59
+ "FaultProxy",
60
+ "Invocation",
61
+ "LLMTarget",
62
+ "MCPTarget",
63
+ "MockTarget",
64
+ "PHASES",
65
+ "Policy",
66
+ "PolicyError",
67
+ "Probe",
68
+ "ProbeConfig",
69
+ "ProbeResult",
70
+ "Request",
71
+ "Response",
72
+ "ScanResult",
73
+ "Target",
74
+ "TargetInfo",
75
+ "ToolInfo",
76
+ "Trajectory",
77
+ "__version__",
78
+ "available_probes",
79
+ "build_target",
80
+ "get_probe",
81
+ "scan",
82
+ ]
ratemyagent/cli.py ADDED
@@ -0,0 +1,461 @@
1
+ """click CLI: `ratemyagent scan ...`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import json
7
+ import logging
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ import click
12
+
13
+ from . import __version__
14
+ from .models import ScanResult
15
+ from .outputs import render_report, render_scorecard, write_agents_md
16
+ from .outputs.agents_md import applicable_advice
17
+ from .policy import DEFAULT_POLICY_PATH, Policy, PolicyError
18
+ from .probes import PHASES, PLANNED, ProbeConfig, available_probes, resolve_phases, resolve_probes
19
+ from .scanner import scan as run_scan
20
+ from .targets import TargetError, build_target
21
+
22
+ CONTEXT_SETTINGS = {"help_option_names": ["-h", "--help"], "max_content_width": 100}
23
+
24
+ IMPLEMENTED_OUTPUTS = frozenset({"scorecard", "report", "agents-md"})
25
+ PLANNED_OUTPUTS: dict[str, str] = {}
26
+
27
+
28
+ @click.group(context_settings=CONTEXT_SETTINGS)
29
+ @click.version_option(version=__version__, prog_name="ratemyagent")
30
+ def cli() -> None:
31
+ """SRE reliability scanner for AI agents, MCP servers, and LLM tools."""
32
+
33
+
34
+ @cli.command(context_settings=CONTEXT_SETTINGS)
35
+ @click.option(
36
+ "--target",
37
+ "target_kind",
38
+ type=click.Choice(["mcp", "llm", "mock"]),
39
+ required=True,
40
+ help="What to scan. 'mock' runs against a built-in synthetic target, no server needed.",
41
+ )
42
+ @click.option("--uri", help="MCP endpoint: stdio://./server.py or sse://host:port/sse")
43
+ @click.option("--provider", type=click.Choice(["anthropic", "openai"]), help="LLM provider.")
44
+ @click.option("--model", help="LLM model id.")
45
+ @click.option("--tool", help="MCP tool to probe. Defaults to the first tool discovered.")
46
+ @click.option("--tool-args", help="JSON object of arguments for --tool.")
47
+ @click.option(
48
+ "--profile",
49
+ type=click.Choice(["healthy", "degraded", "failing", "saturating", "bloated"]),
50
+ default="healthy",
51
+ show_default=True,
52
+ help="Behavior of the mock target.",
53
+ )
54
+ @click.option("--price-in", type=float,
55
+ help="USD per 1M input tokens, overriding the built-in price table.")
56
+ @click.option("--price-out", type=float,
57
+ help="USD per 1M output tokens, overriding the built-in price table.")
58
+ @click.option(
59
+ "--probes",
60
+ "probe_spec",
61
+ default="all",
62
+ show_default=True,
63
+ help=f"Comma-separated probes, or 'all'. Available: {', '.join(available_probes())}.",
64
+ )
65
+ @click.option(
66
+ "--phases",
67
+ "phase_spec",
68
+ default="all",
69
+ show_default=True,
70
+ help=f"Comma-separated pipeline phases, or 'all'. Order is fixed: {', '.join(PHASES)}.",
71
+ )
72
+ @click.option(
73
+ "--fault-rate",
74
+ type=float,
75
+ default=0.2,
76
+ show_default=True,
77
+ help="Share of calls the chaos phase faults, spread across all fault kinds.",
78
+ )
79
+ @click.option(
80
+ "--output",
81
+ type=click.Choice(["scorecard", "report", "agents-md", "all"]),
82
+ default="scorecard",
83
+ show_default=True,
84
+ help="Output format.",
85
+ )
86
+ @click.option("--requests", "request_count", type=int, default=20, show_default=True,
87
+ help="Requests per probe.")
88
+ @click.option("--concurrency", type=int, default=5, show_default=True,
89
+ help="Max concurrent requests for the load tester.")
90
+ @click.option("--timeout", type=float, default=30.0, show_default=True,
91
+ help="Per-request timeout in seconds.")
92
+ @click.option("--warmup", type=int, default=1, show_default=True,
93
+ help="Unmeasured requests sent before profiling.")
94
+ @click.option("--seed", type=int, default=1337, show_default=True,
95
+ help="Seed for reproducible runs.")
96
+ @click.option("--policy", "policy_path", type=click.Path(dir_okay=False, path_type=Path),
97
+ help="Reliability policy YAML. Defaults to the shipped production-default.")
98
+ @click.option("--json-out", type=click.Path(dir_okay=False, path_type=Path),
99
+ help="Also write the full result as JSON. Name it *.scan.json to keep it "
100
+ "out of git.")
101
+ @click.option("--report-out", type=click.Path(dir_okay=False, path_type=Path),
102
+ default="REPORT.md", show_default=True,
103
+ help="Where --output report writes the markdown report.")
104
+ @click.option("--agents-md-out", type=click.Path(dir_okay=False, path_type=Path),
105
+ default="AGENTS.md", show_default=True,
106
+ help="Where --output agents-md writes. An existing file is diffed "
107
+ "against, so the guide reports what changed.")
108
+ @click.option("-v", "--verbose", is_flag=True, help="Debug logging.")
109
+ def scan(
110
+ target_kind: str,
111
+ uri: str | None,
112
+ provider: str | None,
113
+ model: str | None,
114
+ tool: str | None,
115
+ tool_args: str | None,
116
+ profile: str,
117
+ price_in: float | None,
118
+ price_out: float | None,
119
+ probe_spec: str,
120
+ phase_spec: str,
121
+ fault_rate: float,
122
+ output: str,
123
+ request_count: int,
124
+ concurrency: int,
125
+ timeout: float,
126
+ warmup: int,
127
+ seed: int,
128
+ policy_path: Path | None,
129
+ json_out: Path | None,
130
+ report_out: Path,
131
+ agents_md_out: Path,
132
+ verbose: bool,
133
+ ) -> None:
134
+ """Scan a target and report on it.
135
+
136
+ \b
137
+ Examples:
138
+ ratemyagent scan --target mock --profile degraded
139
+ ratemyagent scan --target mcp --uri stdio://./server.py
140
+ ratemyagent scan --target mcp --uri stdio://./server.py --probes latency --requests 100
141
+ ratemyagent scan --target mock --output agents-md
142
+ ratemyagent scan --target mock --output all
143
+ """
144
+ _configure_logging(verbose)
145
+
146
+ if target_kind == "llm" and not provider:
147
+ raise click.UsageError(
148
+ "--target llm needs --provider anthropic or --provider openai"
149
+ )
150
+ if target_kind == "mcp" and not uri:
151
+ raise click.UsageError("--target mcp needs --uri, e.g. --uri stdio://./server.py")
152
+ if request_count < 1:
153
+ raise click.UsageError("--requests must be at least 1")
154
+ if warmup < 0:
155
+ raise click.UsageError("--warmup cannot be negative")
156
+ if not 0.0 <= fault_rate <= 1.0:
157
+ raise click.UsageError("--fault-rate must be between 0 and 1")
158
+
159
+ formats = set(IMPLEMENTED_OUTPUTS) if output == "all" else {output}
160
+ unsupported = formats - IMPLEMENTED_OUTPUTS
161
+ if unsupported:
162
+ name = sorted(unsupported)[0]
163
+ raise click.UsageError(f"--output {name} is not implemented yet: {PLANNED_OUTPUTS[name]}")
164
+
165
+ try:
166
+ probes = resolve_probes(probe_spec)
167
+ phases = resolve_phases(phase_spec)
168
+ except KeyError as exc:
169
+ raise click.UsageError(str(exc).strip("'")) from exc
170
+
171
+ policy = _load_policy(policy_path)
172
+
173
+ try:
174
+ target = build_target(
175
+ target_kind,
176
+ uri=uri,
177
+ tool=tool,
178
+ tool_args=_parse_tool_args(tool_args),
179
+ timeout_s=timeout,
180
+ profile=profile,
181
+ provider=provider,
182
+ model=model,
183
+ seed=seed,
184
+ )
185
+ except TargetError as exc:
186
+ raise click.UsageError(str(exc)) from exc
187
+
188
+ config = ProbeConfig(
189
+ requests=request_count,
190
+ concurrency=concurrency,
191
+ timeout_s=timeout,
192
+ warmup=warmup,
193
+ seed=seed,
194
+ extra={
195
+ "fault_rate": fault_rate,
196
+ "model": model,
197
+ "price_in": price_in,
198
+ "price_out": price_out,
199
+ },
200
+ )
201
+
202
+ try:
203
+ result = asyncio.run(
204
+ run_scan(target, probes=probes, phases=phases, config=config, policy=policy)
205
+ )
206
+ except TargetError as exc:
207
+ raise click.ClickException(str(exc)) from exc
208
+
209
+ if "scorecard" in formats:
210
+ # The hint sits inside the scorecard so the verdict stays the last two
211
+ # lines; printing it afterwards would displace what CI greps for.
212
+ hint = None if "agents-md" in formats else _agents_md_hint(result)
213
+ click.echo(render_scorecard(result, hint=hint), nl=False)
214
+
215
+ if "report" in formats:
216
+ _write_text(render_report(result), report_out)
217
+ click.echo(f"Report written to {report_out}")
218
+
219
+ if "agents-md" in formats:
220
+ # Diffs against whatever is already there, so a re-scan reports movement.
221
+ write_agents_md(result, agents_md_out)
222
+ click.echo(_agents_md_summary(result, agents_md_out))
223
+
224
+ if json_out:
225
+ _write_json(result, json_out)
226
+ click.echo(f"Wrote {json_out}")
227
+
228
+
229
+ @cli.command("ci", context_settings=CONTEXT_SETTINGS)
230
+ @click.option(
231
+ "--target", "target_kind", type=click.Choice(["mcp", "llm", "mock"]), required=True,
232
+ help="What to scan.",
233
+ )
234
+ @click.option("--uri", help="MCP endpoint: stdio://./server.py or sse://host:port/sse")
235
+ @click.option("--provider", type=click.Choice(["anthropic", "openai"]), help="LLM provider.")
236
+ @click.option("--model", help="LLM model id.")
237
+ @click.option("--tool", help="MCP tool to probe.")
238
+ @click.option(
239
+ "--profile",
240
+ type=click.Choice(["healthy", "degraded", "failing", "saturating", "bloated"]),
241
+ default="healthy", show_default=True, help="Behavior of the mock target.",
242
+ )
243
+ @click.option("--policy", "policy_path", type=click.Path(dir_okay=False, path_type=Path),
244
+ help="Reliability policy YAML. Defaults to the shipped production-default.")
245
+ @click.option("--requests", "request_count", type=int, default=20, show_default=True)
246
+ @click.option("--concurrency", type=int, default=5, show_default=True)
247
+ @click.option("--timeout", type=float, default=30.0, show_default=True)
248
+ @click.option("--fault-rate", type=float, default=0.2, show_default=True)
249
+ @click.option("--seed", type=int, default=1337, show_default=True)
250
+ @click.option("--price-in", type=float, help="USD per 1M input tokens.")
251
+ @click.option("--price-out", type=float, help="USD per 1M output tokens.")
252
+ @click.option("--json-out", type=click.Path(dir_okay=False, path_type=Path),
253
+ help="Also write the full result as JSON.")
254
+ @click.option("--quiet", is_flag=True, help="Print only the verdict line.")
255
+ @click.option("-v", "--verbose", is_flag=True, help="Debug logging.")
256
+ def ci(
257
+ target_kind: str,
258
+ uri: str | None,
259
+ provider: str | None,
260
+ model: str | None,
261
+ tool: str | None,
262
+ profile: str,
263
+ policy_path: Path | None,
264
+ request_count: int,
265
+ concurrency: int,
266
+ timeout: float,
267
+ fault_rate: float,
268
+ seed: int,
269
+ price_in: float | None,
270
+ price_out: float | None,
271
+ json_out: Path | None,
272
+ quiet: bool,
273
+ verbose: bool,
274
+ ) -> None:
275
+ """Run a full scan and exit non-zero if it misses the policy.
276
+
277
+ Exit codes: 0 the score met pass_score, 1 it did not, 2 the scan could not
278
+ run at all. A gate that cannot distinguish "your agent regressed" from "the
279
+ scanner broke" is not a gate worth having in a pipeline.
280
+
281
+ \b
282
+ Examples:
283
+ ratemyagent ci --target mock --profile healthy
284
+ ratemyagent ci --target mcp --uri stdio://./server.py --policy production.yaml
285
+ """
286
+ _configure_logging(verbose)
287
+
288
+ if target_kind == "llm" and not provider:
289
+ raise click.UsageError("--target llm needs --provider anthropic or --provider openai")
290
+ if target_kind == "mcp" and not uri:
291
+ raise click.UsageError("--target mcp needs --uri, e.g. --uri stdio://./server.py")
292
+
293
+ policy = _load_policy(policy_path)
294
+
295
+ try:
296
+ target = build_target(
297
+ target_kind, uri=uri, tool=tool, timeout_s=timeout, profile=profile,
298
+ provider=provider, model=model, seed=seed,
299
+ )
300
+ config = ProbeConfig(
301
+ requests=request_count, concurrency=concurrency, timeout_s=timeout,
302
+ seed=seed,
303
+ extra={
304
+ "fault_rate": fault_rate, "model": model,
305
+ "price_in": price_in, "price_out": price_out,
306
+ },
307
+ )
308
+ result = asyncio.run(run_scan(target, config=config, policy=policy))
309
+ except (TargetError, PolicyError) as exc:
310
+ # Exit 2: the scan never happened, which is not the same as a failing
311
+ # target and should not be reported as one.
312
+ click.echo(f"error: {exc}", err=True)
313
+ raise SystemExit(2) from exc
314
+
315
+ if not quiet:
316
+ click.echo(render_scorecard(result), nl=False)
317
+
318
+ if result.score is None:
319
+ click.echo(
320
+ f"FAIL no policy threshold in {policy.name} could be evaluated against "
321
+ "this scan",
322
+ err=True,
323
+ )
324
+ raise SystemExit(1)
325
+
326
+ verdict = "PASS" if result.passed else "FAIL"
327
+ click.echo(
328
+ f"{verdict} score {result.score:.1f}/100 "
329
+ f"(policy {policy.name} requires {policy.pass_score:g})"
330
+ )
331
+
332
+ if not result.passed:
333
+ for check in result.failed_checks:
334
+ click.echo(f" failed: {check.name} -- {check.reason}", err=True)
335
+
336
+ if json_out:
337
+ _write_json(result, json_out)
338
+
339
+ raise SystemExit(0 if result.passed else 1)
340
+
341
+
342
+ @cli.command("policy")
343
+ @click.option("--policy", "policy_path", type=click.Path(dir_okay=False, path_type=Path),
344
+ help="Policy YAML to show. Defaults to the shipped production-default.")
345
+ def show_policy(policy_path: Path | None) -> None:
346
+ """Show a policy and what each threshold reads."""
347
+ from .policy import SPECS_BY_NAME
348
+
349
+ policy = _load_policy(policy_path)
350
+ click.echo(f"{policy.name} (pass_score {policy.pass_score:g})")
351
+ if policy.description:
352
+ click.echo(f" {policy.description.strip()}")
353
+ click.echo("\nThresholds:")
354
+ for spec in policy.specs:
355
+ value = policy.thresholds[spec.name]
356
+ click.echo(
357
+ f" {spec.name:<32} {value:<10g} {spec.direction:<4} "
358
+ f"<- {spec.probe}.{spec.metric}"
359
+ )
360
+
361
+ unset = [name for name in SPECS_BY_NAME if name not in policy.thresholds]
362
+ if unset:
363
+ click.echo("\nNot set (not scored):")
364
+ for name in unset:
365
+ click.echo(f" {name}")
366
+
367
+
368
+ @cli.command("probes")
369
+ def list_probes() -> None:
370
+ """List the probes this build can run."""
371
+ click.echo("Available:")
372
+ for name in available_probes():
373
+ from .probes import get_probe
374
+
375
+ click.echo(f" {name:<12} {get_probe(name).description}")
376
+
377
+ if PLANNED:
378
+ click.echo("\nPlanned:")
379
+ for name, note in PLANNED.items():
380
+ click.echo(f" {name:<12} {note}")
381
+
382
+
383
+ def _agents_md_summary(result: ScanResult, path: Path) -> str:
384
+ """What was written, and how much of it needs attention first."""
385
+ sections = applicable_advice(result)
386
+ critical = sum(1 for advice in sections if advice.critical)
387
+
388
+ detail = f"{len(sections)} recommendation{'' if len(sections) == 1 else 's'}"
389
+ if critical:
390
+ detail += f", {critical} critical"
391
+ return f"AGENTS.md written to {path} ({detail})"
392
+
393
+
394
+ def _agents_md_hint(result: ScanResult) -> str | None:
395
+ """Point at the fix guide when there is something to fix.
396
+
397
+ Printed rather than prompted: an SRE tool has to stay pipeable and
398
+ non-blocking, so this never asks a question.
399
+ """
400
+ findings = sum(len(probe.findings) for probe in result.probes)
401
+ probes = sum(1 for probe in result.probes if probe.findings)
402
+ if not findings:
403
+ return None
404
+
405
+ # Only point at the guide when it would actually have something to say. A
406
+ # clean scan still emits informational findings, and sending someone to a
407
+ # file that reads "nothing to fix" wastes their time.
408
+ if not applicable_advice(result):
409
+ return None
410
+
411
+ return (
412
+ f"{findings} finding{'' if findings == 1 else 's'} across "
413
+ f"{probes} probe{'' if probes == 1 else 's'}. "
414
+ "Run with --output agents-md to generate a fix guide."
415
+ )
416
+
417
+
418
+ def _load_policy(path: Path | None) -> Policy:
419
+ """Load a policy file, or the shipped default when none is given."""
420
+ try:
421
+ return Policy.load(path or DEFAULT_POLICY_PATH)
422
+ except PolicyError as exc:
423
+ raise click.UsageError(str(exc)) from exc
424
+
425
+
426
+ def _parse_tool_args(raw: str | None) -> dict[str, Any] | None:
427
+ if raw is None:
428
+ return None
429
+ try:
430
+ parsed = json.loads(raw)
431
+ except json.JSONDecodeError as exc:
432
+ raise click.UsageError(f"--tool-args must be valid JSON: {exc}") from exc
433
+ if not isinstance(parsed, dict):
434
+ raise click.UsageError("--tool-args must be a JSON object")
435
+ return parsed
436
+
437
+
438
+ def _write_text(document: str, path: Path) -> None:
439
+ path.parent.mkdir(parents=True, exist_ok=True)
440
+ path.write_text(document, encoding="utf-8")
441
+
442
+
443
+ def _write_json(result: ScanResult, path: Path) -> None:
444
+ path.parent.mkdir(parents=True, exist_ok=True)
445
+ path.write_text(json.dumps(result.to_dict(), indent=2) + "\n", encoding="utf-8")
446
+
447
+
448
+ def _configure_logging(verbose: bool) -> None:
449
+ logging.basicConfig(
450
+ level=logging.DEBUG if verbose else logging.WARNING,
451
+ format="%(levelname)s %(name)s: %(message)s",
452
+ )
453
+
454
+
455
+ def main() -> None:
456
+ """Console script entry point."""
457
+ cli()
458
+
459
+
460
+ if __name__ == "__main__": # pragma: no cover
461
+ main()