@cleocode/skills 2026.5.82 → 2026.5.83

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/README.md +0 -1
  2. package/package.json +1 -1
  3. package/profiles/recommended.json +1 -1
  4. package/skills/manifest.json +45 -1
  5. package/skills/ct-grade-v2-1/MIGRATION.md +0 -28
  6. package/skills/ct-grade-v2-1/SKILL.md +0 -235
  7. package/skills/ct-grade-v2-1/agents/analysis-reporter.md +0 -203
  8. package/skills/ct-grade-v2-1/agents/blind-comparator.md +0 -157
  9. package/skills/ct-grade-v2-1/agents/scenario-runner.md +0 -160
  10. package/skills/ct-grade-v2-1/evals/evals.json +0 -74
  11. package/skills/ct-grade-v2-1/grade-viewer/__pycache__/build_op_stats.cpython-314.pyc +0 -0
  12. package/skills/ct-grade-v2-1/grade-viewer/__pycache__/generate_grade_review.cpython-314.pyc +0 -0
  13. package/skills/ct-grade-v2-1/grade-viewer/build_op_stats.py +0 -174
  14. package/skills/ct-grade-v2-1/grade-viewer/eval-analysis.json +0 -41
  15. package/skills/ct-grade-v2-1/grade-viewer/eval-report.md +0 -37
  16. package/skills/ct-grade-v2-1/grade-viewer/generate_grade_review.py +0 -1023
  17. package/skills/ct-grade-v2-1/grade-viewer/generate_grade_viewer.py +0 -548
  18. package/skills/ct-grade-v2-1/grade-viewer/grade-review-eval.html +0 -613
  19. package/skills/ct-grade-v2-1/grade-viewer/grade-review.html +0 -1532
  20. package/skills/ct-grade-v2-1/grade-viewer/viewer.html +0 -620
  21. package/skills/ct-grade-v2-1/manifest-entry.json +0 -31
  22. package/skills/ct-grade-v2-1/references/ab-testing.md +0 -173
  23. package/skills/ct-grade-v2-1/references/domains-ssot.md +0 -156
  24. package/skills/ct-grade-v2-1/references/grade-spec-v2.md +0 -167
  25. package/skills/ct-grade-v2-1/references/playbook-v2.md +0 -325
  26. package/skills/ct-grade-v2-1/references/token-tracking.md +0 -200
  27. package/skills/ct-grade-v2-1/scripts/generate_report.py +0 -419
  28. package/skills/ct-grade-v2-1/scripts/run_ab_test.py +0 -493
  29. package/skills/ct-grade-v2-1/scripts/run_scenario.py +0 -396
  30. package/skills/ct-grade-v2-1/scripts/setup_run.py +0 -207
  31. package/skills/ct-grade-v2-1/scripts/token_tracker.py +0 -175
@@ -1,175 +0,0 @@
1
- #!/usr/bin/env python3
2
- """
3
- token_tracker.py — Aggregate token usage stats from a completed A/B run.
4
-
5
- Usage:
6
- python token_tracker.py --run-dir ./ab_results/run-001
7
-
8
- Reads all timing.json files in the run directory and produces token-summary.json
9
- with per-arm statistics.
10
-
11
- Output: <run-dir>/token-summary.json
12
- """
13
-
14
- import argparse
15
- import json
16
- import os
17
- import sys
18
- import math
19
- from pathlib import Path
20
-
21
-
22
- def find_timing_files(run_dir):
23
- """Find all timing.json files under run_dir."""
24
- return list(Path(run_dir).rglob("timing.json"))
25
-
26
-
27
- def load_timing(path):
28
- try:
29
- with open(path) as f:
30
- return json.load(f)
31
- except Exception as e:
32
- print(f" WARN: Could not read {path}: {e}", file=sys.stderr)
33
- return None
34
-
35
-
36
- def mean(values):
37
- return sum(values) / len(values) if values else 0
38
-
39
-
40
- def stddev(values):
41
- if len(values) < 2:
42
- return 0
43
- m = mean(values)
44
- return math.sqrt(sum((x - m) ** 2 for x in values) / (len(values) - 1))
45
-
46
-
47
- def stats(values):
48
- if not values:
49
- return {"mean": None, "stddev": None, "min": None, "max": None, "count": 0}
50
- return {
51
- "mean": round(mean(values), 1),
52
- "stddev": round(stddev(values), 1),
53
- "min": min(values),
54
- "max": max(values),
55
- "count": len(values),
56
- }
57
-
58
-
59
- def main():
60
- parser = argparse.ArgumentParser(description="Aggregate token stats from ct-grade A/B run")
61
- parser.add_argument("--run-dir", required=True)
62
- parser.add_argument("--output", default=None, help="Output path (default: <run-dir>/token-summary.json)")
63
- args = parser.parse_args()
64
-
65
- run_dir = args.run_dir
66
- if not os.path.isdir(run_dir):
67
- print(f"ERROR: Run dir not found: {run_dir}", file=sys.stderr)
68
- sys.exit(1)
69
-
70
- timing_files = find_timing_files(run_dir)
71
- if not timing_files:
72
- print(f"ERROR: No timing.json files found in {run_dir}", file=sys.stderr)
73
- sys.exit(1)
74
-
75
- print(f"Found {len(timing_files)} timing.json files")
76
-
77
- # Group by arm
78
- by_arm = {}
79
- by_interface = {}
80
- missing_tokens = []
81
-
82
- for tpath in timing_files:
83
- data = load_timing(tpath)
84
- if data is None:
85
- continue
86
-
87
- arm = data.get("arm", "unknown")
88
- iface = data.get("interface", "unknown")
89
- tokens = data.get("total_tokens")
90
- duration = data.get("duration_ms")
91
-
92
- if arm not in by_arm:
93
- by_arm[arm] = {"tokens": [], "duration_ms": [], "interface": iface, "files": []}
94
- if iface not in by_interface:
95
- by_interface[iface] = {"tokens": [], "duration_ms": [], "files": []}
96
-
97
- by_arm[arm]["files"].append(str(tpath))
98
-
99
- if tokens is not None:
100
- by_arm[arm]["tokens"].append(tokens)
101
- by_interface[iface]["tokens"].append(tokens)
102
- else:
103
- missing_tokens.append(str(tpath))
104
-
105
- if duration is not None:
106
- by_arm[arm]["duration_ms"].append(duration)
107
- by_interface[iface]["duration_ms"].append(duration)
108
-
109
- # Build summary
110
- arm_stats = {}
111
- for arm, data in sorted(by_arm.items()):
112
- arm_stats[arm] = {
113
- "interface": data["interface"],
114
- "file_count": len(data["files"]),
115
- "total_tokens": stats(data["tokens"]),
116
- "duration_ms": stats(data["duration_ms"]),
117
- }
118
-
119
- iface_stats = {}
120
- for iface, data in sorted(by_interface.items()):
121
- iface_stats[iface] = {
122
- "file_count": len(data["files"]),
123
- "total_tokens": stats(data["tokens"]),
124
- "duration_ms": stats(data["duration_ms"]),
125
- }
126
-
127
- # Compute delta between arms (A vs B)
128
- delta = {}
129
- if "arm-A" in arm_stats and "arm-B" in arm_stats:
130
- a_mean = arm_stats["arm-A"]["total_tokens"].get("mean") or 0
131
- b_mean = arm_stats["arm-B"]["total_tokens"].get("mean") or 0
132
- if b_mean > 0:
133
- delta = {
134
- "mean_tokens": round(a_mean - b_mean, 1),
135
- "percent": f"{((a_mean - b_mean) / b_mean * 100):+.1f}%",
136
- "note": f"Arm A uses {abs(a_mean - b_mean):.0f} {'more' if a_mean > b_mean else 'fewer'} tokens on average",
137
- }
138
-
139
- summary = {
140
- "run_dir": os.path.abspath(run_dir),
141
- "timing_files_found": len(timing_files),
142
- "timing_files_missing_tokens": len(missing_tokens),
143
- "by_arm": arm_stats,
144
- "by_interface": iface_stats,
145
- "delta_A_vs_B": delta,
146
- "warnings": (
147
- [f"MISSING total_tokens in {len(missing_tokens)} files — fill these from task notifications"]
148
- if missing_tokens else []
149
- ),
150
- }
151
-
152
- output_path = args.output or os.path.join(run_dir, "token-summary.json")
153
- with open(output_path, "w") as f:
154
- json.dump(summary, f, indent=2)
155
-
156
- # Print summary
157
- print(f"\nToken Summary")
158
- print(f"{'='*50}")
159
- for arm, s in arm_stats.items():
160
- t = s["total_tokens"]
161
- if t["mean"] is not None:
162
- print(f" {arm} ({s['interface']}): {t['mean']:.0f} tokens (±{t['stddev']:.0f}, n={t['count']})")
163
- else:
164
- print(f" {arm} ({s['interface']}): NO TOKEN DATA (fill timing.json from task notifications)")
165
- if delta:
166
- print(f"\n Delta (A-B): {delta['percent']} ({delta['mean_tokens']:+.0f} tokens)")
167
- print(f" {delta['note']}")
168
- if missing_tokens:
169
- print(f"\n WARNING: {len(missing_tokens)} files missing total_tokens")
170
- print(f" These must be filled from Claude Code task notification data.")
171
- print(f"\nWritten: {output_path}")
172
-
173
-
174
- if __name__ == "__main__":
175
- main()