agentskills-tools 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,20 @@
1
+ """Command line tools for authoring and validating Agent Skills.
2
+
3
+ The console script is ``agentskills``. Every command is also reachable
4
+ in-process through :func:`~agentskills_tools.cli.main`, which takes an
5
+ argument list and returns an exit code rather than calling
6
+ :func:`sys.exit` — that is what makes it testable.
7
+ """
8
+
9
+ from agentskills_tools.cli import main
10
+ from agentskills_tools.discovery import CliError, SkillLocation, discover
11
+ from agentskills_tools.findings import Finding, SkillReport
12
+
13
+ __all__ = [
14
+ "CliError",
15
+ "Finding",
16
+ "SkillLocation",
17
+ "SkillReport",
18
+ "discover",
19
+ "main",
20
+ ]
@@ -0,0 +1,10 @@
1
+ """Entry point for ``python -m agentskills_tools`` and the ``agentskills`` script."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+
7
+ from agentskills_tools.cli import main
8
+
9
+ if __name__ == "__main__":
10
+ sys.exit(main())
@@ -0,0 +1,444 @@
1
+ """Argument parsing and dispatch for the ``agentskills`` command.
2
+
3
+ Exit codes are the contract CI depends on:
4
+
5
+ * ``0`` — the command ran and found nothing wrong.
6
+ * ``1`` — the command ran and found errors (or, under ``--strict``,
7
+ warnings).
8
+ * ``2`` — the command could not run: a bad path, a missing extra, an
9
+ unwritable directory.
10
+
11
+ Keeping "found a problem" distinct from "could not look" means a
12
+ workflow can tell a broken skill from a broken invocation.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import argparse
18
+ import asyncio
19
+ import json
20
+ import logging
21
+ import sys
22
+ from pathlib import Path
23
+ from typing import Any, TextIO
24
+
25
+ from agentskills_core import LOGGER_NAMESPACE, Skill
26
+ from agentskills_fs import LocalFileSystemSkillProvider
27
+ from agentskills_tools.cost import (
28
+ TOKENIZERS,
29
+ SkillCost,
30
+ cost_exit_code,
31
+ cost_payload,
32
+ cost_skill,
33
+ render_cost_text,
34
+ resolve_counter,
35
+ )
36
+ from agentskills_tools.discovery import CliError, SkillLocation, discover, relative_to_cwd
37
+ from agentskills_tools.evals import (
38
+ DEFAULT_CACHE_DIR,
39
+ CompletionCache,
40
+ EvalRunner,
41
+ eval_exit_code,
42
+ load_model,
43
+ render_results_text,
44
+ results_payload,
45
+ run_suites,
46
+ )
47
+ from agentskills_tools.evalspec import EvalSuite, load_skill_evals
48
+ from agentskills_tools.findings import SkillReport
49
+ from agentskills_tools.inspection import inspect_location, render_inspection_text
50
+ from agentskills_tools.lint import DEFAULT_BODY_TOKEN_BUDGET, lint_locations
51
+ from agentskills_tools.render import (
52
+ SCHEMA_VERSION,
53
+ exit_code,
54
+ plural,
55
+ render_github,
56
+ render_json,
57
+ render_text,
58
+ )
59
+ from agentskills_tools.scaffold import DEFAULT_DESCRIPTION, init_from, init_skill
60
+ from agentskills_tools.serve import build_registry, create_server
61
+ from agentskills_tools.validate import validate_locations
62
+
63
+ EXIT_OK = 0
64
+ EXIT_FINDINGS = 1
65
+ EXIT_ERROR = 2
66
+
67
+
68
+ def _format_parent(*choices: str) -> argparse.ArgumentParser:
69
+ """Build a parent parser offering ``--format`` over *choices*.
70
+
71
+ Only ``validate`` and ``lint`` produce findings, so only they can
72
+ offer ``github``; advertising it on ``inspect`` would promise an
73
+ output that command has no way to produce.
74
+ """
75
+ parent = argparse.ArgumentParser(add_help=False)
76
+ parent.add_argument(
77
+ "--format",
78
+ choices=list(choices),
79
+ default="text",
80
+ help="Output format (default: text).",
81
+ )
82
+ return parent
83
+
84
+
85
+ def build_parser() -> argparse.ArgumentParser:
86
+ """Build the ``agentskills`` argument parser."""
87
+ common = argparse.ArgumentParser(add_help=False)
88
+ common.add_argument(
89
+ "-v",
90
+ "--verbose",
91
+ action="store_true",
92
+ help="Print the SDK's debug logs to stderr.",
93
+ )
94
+
95
+ reported = _format_parent("text", "json", "github")
96
+ formatted = _format_parent("text", "json")
97
+
98
+ parser = argparse.ArgumentParser(
99
+ prog="agentskills",
100
+ description="Author, validate, and serve Agent Skills.",
101
+ parents=[common],
102
+ )
103
+ subparsers = parser.add_subparsers(dest="command", required=True)
104
+
105
+ init = subparsers.add_parser(
106
+ "init",
107
+ parents=[common],
108
+ help="Scaffold a new skill.",
109
+ description="Scaffold a new skill directory that already validates.",
110
+ )
111
+ init.add_argument("name", nargs="?", help="Skill name, also used as the directory name.")
112
+ init.add_argument(
113
+ "--path",
114
+ type=Path,
115
+ default=Path("."),
116
+ help="Directory to create the skill in (default: the working directory).",
117
+ )
118
+ init.add_argument(
119
+ "--description",
120
+ default=None,
121
+ help="Frontmatter description for the new skill.",
122
+ )
123
+ init.add_argument(
124
+ "--from",
125
+ dest="source",
126
+ type=Path,
127
+ help="Import an AGENTS.md, Copilot file, Cursor rule, or Claude skill folder.",
128
+ )
129
+
130
+ validate = subparsers.add_parser(
131
+ "validate",
132
+ parents=[common, reported],
133
+ help="Check skills against the specification.",
134
+ description="Check one skill or a directory of skills. Exits 1 on any error.",
135
+ )
136
+ validate.add_argument("path", type=Path, help="A skill folder or a folder of skills.")
137
+
138
+ lint = subparsers.add_parser(
139
+ "lint",
140
+ parents=[common, reported],
141
+ help="Report quality warnings.",
142
+ description="Report problems that are legal per the spec but still cost you.",
143
+ )
144
+ lint.add_argument("path", type=Path, help="A skill folder or a folder of skills.")
145
+ lint.add_argument(
146
+ "--strict",
147
+ action="store_true",
148
+ help="Exit 1 on warnings as well as errors.",
149
+ )
150
+ lint.add_argument(
151
+ "--max-body-tokens",
152
+ type=int,
153
+ default=DEFAULT_BODY_TOKEN_BUDGET,
154
+ metavar="N",
155
+ help=f"Estimated body token budget (default: {DEFAULT_BODY_TOKEN_BUDGET}).",
156
+ )
157
+
158
+ inspect = subparsers.add_parser(
159
+ "inspect",
160
+ parents=[common, formatted],
161
+ help="Show what an agent would receive.",
162
+ description="Render the catalog entry, metadata, and body, with estimated token cost.",
163
+ )
164
+ inspect.add_argument("path", type=Path, help="A skill folder or a folder of skills.")
165
+ inspect.add_argument(
166
+ "--cost",
167
+ action="store_true",
168
+ help="Report token cost per turn, per load, and on demand instead of the content.",
169
+ )
170
+ inspect.add_argument(
171
+ "--budget",
172
+ type=int,
173
+ metavar="N",
174
+ help="With --cost, exit 1 when catalog entry plus body exceeds N tokens.",
175
+ )
176
+ inspect.add_argument(
177
+ "--turn-budget",
178
+ type=int,
179
+ metavar="N",
180
+ help="With --cost, exit 1 when the catalog entry alone exceeds N tokens.",
181
+ )
182
+ inspect.add_argument(
183
+ "--tokenizer",
184
+ choices=list(TOKENIZERS),
185
+ default="auto",
186
+ help="Token counter (default: auto, which uses tiktoken when it is usable).",
187
+ )
188
+
189
+ evaluate = subparsers.add_parser(
190
+ "eval",
191
+ parents=[common, formatted],
192
+ help="Measure what difference a skill makes.",
193
+ description=(
194
+ "Run each eval case twice, with and without the skill, and report "
195
+ "the delta. Calls a real model: opt in deliberately."
196
+ ),
197
+ )
198
+ evaluate.add_argument("path", type=Path, help="A skill folder or a folder of skills.")
199
+ evaluate.add_argument(
200
+ "--model",
201
+ required=True,
202
+ metavar="MODULE:FACTORY",
203
+ help="Dotted path to a zero-argument callable returning a model client.",
204
+ )
205
+ evaluate.add_argument(
206
+ "--judge",
207
+ metavar="MODULE:FACTORY",
208
+ help="Model client for judged expectations (default: the model under test).",
209
+ )
210
+ evaluate.add_argument(
211
+ "--cache-dir",
212
+ type=Path,
213
+ default=DEFAULT_CACHE_DIR,
214
+ help=f"Where to cache completions (default: {DEFAULT_CACHE_DIR}).",
215
+ )
216
+ evaluate.add_argument(
217
+ "--no-cache",
218
+ action="store_true",
219
+ help="Call the model for every attempt, ignoring and not writing the cache.",
220
+ )
221
+
222
+ serve = subparsers.add_parser(
223
+ "serve",
224
+ parents=[common],
225
+ help="Run an MCP server over a folder of skills.",
226
+ description="Run an MCP server over a folder of skills, with no config file.",
227
+ )
228
+ serve.add_argument("path", type=Path, help="A skill folder or a folder of skills.")
229
+ serve.add_argument(
230
+ "--transport",
231
+ choices=["stdio", "streamable-http"],
232
+ default="stdio",
233
+ help="MCP transport (default: stdio).",
234
+ )
235
+ serve.add_argument(
236
+ "--name",
237
+ default="Agent Skills",
238
+ help="Display name advertised by the server.",
239
+ )
240
+
241
+ return parser
242
+
243
+
244
+ def _enable_debug_logging() -> None:
245
+ """Send the SDK's own logs to stderr, leaving stdout parseable."""
246
+ handler = logging.StreamHandler(sys.stderr)
247
+ handler.setFormatter(logging.Formatter("%(levelname)s %(name)s: %(message)s"))
248
+ logger = logging.getLogger(LOGGER_NAMESPACE)
249
+ logger.addHandler(handler)
250
+ logger.setLevel(logging.DEBUG)
251
+
252
+
253
+ def _run_init(args: argparse.Namespace, out: TextIO) -> int:
254
+ if args.source:
255
+ target = asyncio.run(init_from(args.source, args.path, args.name, args.description))
256
+ elif args.name:
257
+ target = asyncio.run(
258
+ init_skill(args.name, args.path, args.description or DEFAULT_DESCRIPTION)
259
+ )
260
+ else:
261
+ raise CliError("init requires a skill name or --from source")
262
+ print(f"Created {relative_to_cwd(target)}", file=out)
263
+ return EXIT_OK
264
+
265
+
266
+ def _render(
267
+ command: str,
268
+ output_format: str,
269
+ reports: list[SkillReport],
270
+ out: TextIO,
271
+ *,
272
+ strict: bool = False,
273
+ ) -> None:
274
+ """Write *reports* in the requested format."""
275
+ if output_format == "json":
276
+ render_json(command, reports, out, strict=strict)
277
+ elif output_format == "github":
278
+ render_github(reports, out)
279
+ else:
280
+ render_text(reports, out)
281
+
282
+
283
+ def _run_validate(args: argparse.Namespace, out: TextIO) -> int:
284
+ root, locations = discover(args.path)
285
+ reports = asyncio.run(validate_locations(root, locations))
286
+ _render("validate", args.format, reports, out)
287
+ return exit_code(reports)
288
+
289
+
290
+ def _run_lint(args: argparse.Namespace, out: TextIO) -> int:
291
+ root, locations = discover(args.path)
292
+ reports = asyncio.run(lint_locations(root, locations, body_token_budget=args.max_body_tokens))
293
+ _render("lint", args.format, reports, out, strict=args.strict)
294
+ return exit_code(reports, strict=args.strict)
295
+
296
+
297
+ async def _inspect_all(root: Path, locations: list[SkillLocation]) -> list[dict[str, Any]]:
298
+ return [await inspect_location(root, location) for location in locations]
299
+
300
+
301
+ async def _cost_all(root: Path, locations: list[SkillLocation], tokenizer: str) -> list[SkillCost]:
302
+ counter = resolve_counter(tokenizer)
303
+ provider = LocalFileSystemSkillProvider(root)
304
+ costs = []
305
+ for location in locations:
306
+ inspection = await inspect_location(root, location)
307
+ skill = Skill(location.skill_id, provider)
308
+ costs.append(
309
+ await cost_skill(
310
+ skill,
311
+ inspection["path"],
312
+ inspection["catalogEntry"],
313
+ inspection["body"],
314
+ counter,
315
+ )
316
+ )
317
+ return costs
318
+
319
+
320
+ def _run_cost(args: argparse.Namespace, out: TextIO) -> int:
321
+ root, locations = discover(args.path)
322
+ costs = asyncio.run(_cost_all(root, locations, args.tokenizer))
323
+ budgets = {"budget": args.budget, "turn_budget": args.turn_budget}
324
+ if args.format == "json":
325
+ payload = {
326
+ "schemaVersion": SCHEMA_VERSION,
327
+ "command": "inspect",
328
+ "skills": cost_payload(costs, **budgets),
329
+ }
330
+ json.dump(payload, out, indent=2)
331
+ print(file=out)
332
+ else:
333
+ render_cost_text(costs, out, **budgets)
334
+ return cost_exit_code(costs, **budgets)
335
+
336
+
337
+ def _run_inspect(args: argparse.Namespace, out: TextIO) -> int:
338
+ if args.cost:
339
+ return _run_cost(args, out)
340
+
341
+ root, locations = discover(args.path)
342
+ inspections = asyncio.run(_inspect_all(root, locations))
343
+ if args.format == "json":
344
+ payload = {
345
+ "schemaVersion": SCHEMA_VERSION,
346
+ "command": "inspect",
347
+ "skills": inspections,
348
+ }
349
+ json.dump(payload, out, indent=2)
350
+ print(file=out)
351
+ else:
352
+ for index, inspection in enumerate(inspections):
353
+ if index:
354
+ print("\n" + "-" * 60 + "\n", file=out)
355
+ render_inspection_text(inspection, out)
356
+ return EXIT_OK
357
+
358
+
359
+ async def _collect_suites(
360
+ root: Path, locations: list[SkillLocation]
361
+ ) -> tuple[list[tuple[EvalSuite, str]], list[SkillReport]]:
362
+ """Load every eval suite, pairing each with its skill's body."""
363
+ provider = LocalFileSystemSkillProvider(root)
364
+ pairs: list[tuple[EvalSuite, str]] = []
365
+ broken: list[SkillReport] = []
366
+ for location in locations:
367
+ suites, findings = load_skill_evals(location.path, location.skill_id)
368
+ if findings:
369
+ broken.append(SkillReport(location.skill_id, location.path, findings))
370
+ if not suites:
371
+ continue
372
+ # The skill's body is the whole intervention being measured, so
373
+ # it is what goes in the system prompt for the "with" run.
374
+ body = await Skill(location.skill_id, provider).get_body()
375
+ pairs.extend((suite, body) for suite in suites)
376
+ return pairs, broken
377
+
378
+
379
+ def _run_eval(args: argparse.Namespace, out: TextIO) -> int:
380
+ root, locations = discover(args.path)
381
+ pairs, broken = asyncio.run(_collect_suites(root, locations))
382
+ if broken:
383
+ render_text(broken, sys.stderr)
384
+ print(
385
+ "error: fix the eval files above before spending money on them",
386
+ file=sys.stderr,
387
+ )
388
+ return EXIT_ERROR
389
+ if not pairs:
390
+ raise CliError(f"no eval cases found under {relative_to_cwd(root)}")
391
+
392
+ model = load_model(args.model)
393
+ judge = load_model(args.judge) if args.judge else None
394
+ cache = CompletionCache(None if args.no_cache else args.cache_dir)
395
+ runner = EvalRunner(model, judge=judge, cache=cache)
396
+ results = asyncio.run(run_suites(runner, pairs))
397
+
398
+ if args.format == "json":
399
+ payload = {"schemaVersion": SCHEMA_VERSION, **results_payload(results)}
400
+ json.dump(payload, out, indent=2)
401
+ print(file=out)
402
+ else:
403
+ render_results_text(results, out)
404
+ return eval_exit_code(results)
405
+
406
+
407
+ def _run_serve(args: argparse.Namespace, out: TextIO) -> int:
408
+ root, locations = discover(args.path)
409
+ registry = asyncio.run(build_registry(root, locations))
410
+ server = create_server(registry, name=args.name)
411
+ print(f"Serving {plural(len(locations), 'skill')} over {args.transport}", file=sys.stderr)
412
+ server.run(transport=args.transport)
413
+ return EXIT_OK
414
+
415
+
416
+ _COMMANDS = {
417
+ "init": _run_init,
418
+ "validate": _run_validate,
419
+ "lint": _run_lint,
420
+ "inspect": _run_inspect,
421
+ "eval": _run_eval,
422
+ "serve": _run_serve,
423
+ }
424
+
425
+
426
+ def main(argv: list[str] | None = None, out: TextIO | None = None) -> int:
427
+ """Run the ``agentskills`` command line.
428
+
429
+ Args:
430
+ argv: Arguments to parse. Defaults to ``sys.argv[1:]``.
431
+ out: Stream to write reports to. Defaults to stdout.
432
+
433
+ Returns:
434
+ A process exit code.
435
+ """
436
+ args = build_parser().parse_args(argv)
437
+ if args.verbose:
438
+ _enable_debug_logging()
439
+
440
+ try:
441
+ return _COMMANDS[args.command](args, out or sys.stdout)
442
+ except CliError as exc:
443
+ print(f"error: {exc}", file=sys.stderr)
444
+ return EXIT_ERROR