@cleocode/skills 2026.5.82 → 2026.5.84

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +0 -1
  2. package/package.json +1 -1
  3. package/profiles/recommended.json +1 -1
  4. package/skills/_shared/__tests__/lifecycle-protocol-reconcile.test.ts +112 -0
  5. package/skills/_shared/__tests__/loom-adr-links.test.ts +163 -0
  6. package/skills/_shared/__tests__/loom-stage-coverage.test.ts +167 -0
  7. package/skills/ct-adr-recorder/SKILL.md +18 -0
  8. package/skills/ct-consensus-voter/SKILL.md +14 -0
  9. package/skills/ct-contribution/SKILL.md +80 -0
  10. package/skills/ct-epic-architect/SKILL.md +15 -0
  11. package/skills/ct-ivt-looper/SKILL.md +32 -0
  12. package/skills/ct-release-orchestrator/SKILL.md +16 -0
  13. package/skills/ct-research-agent/SKILL.md +15 -0
  14. package/skills/ct-spec-writer/SKILL.md +15 -0
  15. package/skills/ct-task-executor/SKILL.md +15 -0
  16. package/skills/ct-validator/SKILL.md +35 -0
  17. package/skills/manifest.json +81 -9
  18. package/skills/ct-grade-v2-1/MIGRATION.md +0 -28
  19. package/skills/ct-grade-v2-1/SKILL.md +0 -235
  20. package/skills/ct-grade-v2-1/agents/analysis-reporter.md +0 -203
  21. package/skills/ct-grade-v2-1/agents/blind-comparator.md +0 -157
  22. package/skills/ct-grade-v2-1/agents/scenario-runner.md +0 -160
  23. package/skills/ct-grade-v2-1/evals/evals.json +0 -74
  24. package/skills/ct-grade-v2-1/grade-viewer/__pycache__/build_op_stats.cpython-314.pyc +0 -0
  25. package/skills/ct-grade-v2-1/grade-viewer/__pycache__/generate_grade_review.cpython-314.pyc +0 -0
  26. package/skills/ct-grade-v2-1/grade-viewer/build_op_stats.py +0 -174
  27. package/skills/ct-grade-v2-1/grade-viewer/eval-analysis.json +0 -41
  28. package/skills/ct-grade-v2-1/grade-viewer/eval-report.md +0 -37
  29. package/skills/ct-grade-v2-1/grade-viewer/generate_grade_review.py +0 -1023
  30. package/skills/ct-grade-v2-1/grade-viewer/generate_grade_viewer.py +0 -548
  31. package/skills/ct-grade-v2-1/grade-viewer/grade-review-eval.html +0 -613
  32. package/skills/ct-grade-v2-1/grade-viewer/grade-review.html +0 -1532
  33. package/skills/ct-grade-v2-1/grade-viewer/viewer.html +0 -620
  34. package/skills/ct-grade-v2-1/manifest-entry.json +0 -31
  35. package/skills/ct-grade-v2-1/references/ab-testing.md +0 -173
  36. package/skills/ct-grade-v2-1/references/domains-ssot.md +0 -156
  37. package/skills/ct-grade-v2-1/references/grade-spec-v2.md +0 -167
  38. package/skills/ct-grade-v2-1/references/playbook-v2.md +0 -325
  39. package/skills/ct-grade-v2-1/references/token-tracking.md +0 -200
  40. package/skills/ct-grade-v2-1/scripts/generate_report.py +0 -419
  41. package/skills/ct-grade-v2-1/scripts/run_ab_test.py +0 -493
  42. package/skills/ct-grade-v2-1/scripts/run_scenario.py +0 -396
  43. package/skills/ct-grade-v2-1/scripts/setup_run.py +0 -207
  44. package/skills/ct-grade-v2-1/scripts/token_tracker.py +0 -175
@@ -1,174 +0,0 @@
1
- #!/usr/bin/env python3
2
- """
3
- build_op_stats.py — Aggregate operations.jsonl files from grade runs into per-operation statistics.
4
-
5
- Reads all operations.jsonl files under --grade-runs-dir and computes per-operation stats
6
- split by interface (mcp/cli). Output is a JSON object keyed by "domain.operation".
7
-
8
- Usage:
9
- python build_op_stats.py [options]
10
-
11
- Options:
12
- --grade-runs-dir PATH Directory containing grade run subdirectories
13
- (default: .cleo/metrics/grade-runs relative to cwd)
14
- --output PATH Output JSON file path
15
- (default: .cleo/metrics/per_operation_stats.json)
16
- --pretty Pretty-print JSON output (default: compact)
17
- --verbose Print progress to stderr
18
-
19
- Output format (per key "domain.operation"):
20
- {
21
- "mcp_calls": 42,
22
- "cli_calls": 10,
23
- "total_mcp_ms": 1234.5,
24
- "total_cli_ms": 456.7,
25
- "avg_mcp_ms": 29.4,
26
- "avg_cli_ms": 45.7,
27
- "runs_seen": 3
28
- }
29
-
30
- Also importable as a module:
31
- from build_op_stats import compute_stats
32
- stats = compute_stats(grade_runs_dir="/path/to/grade-runs")
33
- """
34
-
35
- import argparse
36
- import json
37
- import sys
38
- from pathlib import Path
39
-
40
-
41
- def compute_stats(grade_runs_dir, verbose=False):
42
- """
43
- Aggregate operations.jsonl files under grade_runs_dir.
44
-
45
- Returns dict keyed by "domain.operation" with accumulated stats.
46
- """
47
- runs_dir = Path(grade_runs_dir)
48
- stats = {}
49
- files_processed = 0
50
- lines_processed = 0
51
-
52
- if not runs_dir.exists():
53
- if verbose:
54
- print(f"[build_op_stats] Grade runs dir not found: {runs_dir}", file=sys.stderr)
55
- return stats
56
-
57
- for ops_file in sorted(runs_dir.rglob('operations.jsonl')):
58
- files_processed += 1
59
- if verbose:
60
- print(f"[build_op_stats] Processing: {ops_file}", file=sys.stderr)
61
-
62
- for line in ops_file.read_text(errors='replace').splitlines():
63
- line = line.strip()
64
- if not line:
65
- continue
66
- try:
67
- entry = json.loads(line)
68
- except json.JSONDecodeError:
69
- continue
70
-
71
- domain = entry.get('domain', 'unknown')
72
- operation = entry.get('operation', 'unknown')
73
- key = f"{domain}.{operation}"
74
- interface = entry.get('interface', 'mcp')
75
- duration = float(entry.get('duration_ms', 0) or 0)
76
-
77
- if key not in stats:
78
- stats[key] = {
79
- 'mcp_calls': 0,
80
- 'cli_calls': 0,
81
- 'total_mcp_ms': 0.0,
82
- 'total_cli_ms': 0.0,
83
- 'avg_mcp_ms': 0.0,
84
- 'avg_cli_ms': 0.0,
85
- 'runs_seen': set(),
86
- }
87
-
88
- # Track which run directory this came from
89
- # ops_file is e.g. .../grade-runs/run-20260308/s1/run-01/arm-mcp/operations.jsonl
90
- # run_id is the first path component relative to runs_dir (e.g. "run-20260308")
91
- run_id = ops_file.relative_to(runs_dir).parts[0]
92
- stats[key]['runs_seen'].add(run_id)
93
-
94
- if interface == 'cli':
95
- stats[key]['cli_calls'] += 1
96
- stats[key]['total_cli_ms'] += duration
97
- else:
98
- stats[key]['mcp_calls'] += 1
99
- stats[key]['total_mcp_ms'] += duration
100
-
101
- lines_processed += 1
102
-
103
- # Compute averages and convert sets to counts
104
- for key, v in stats.items():
105
- mc = v['mcp_calls']
106
- cc = v['cli_calls']
107
- v['avg_mcp_ms'] = round(v['total_mcp_ms'] / mc, 2) if mc > 0 else 0.0
108
- v['avg_cli_ms'] = round(v['total_cli_ms'] / cc, 2) if cc > 0 else 0.0
109
- v['total_mcp_ms'] = round(v['total_mcp_ms'], 2)
110
- v['total_cli_ms'] = round(v['total_cli_ms'], 2)
111
- v['runs_seen'] = len(v['runs_seen'])
112
-
113
- if verbose:
114
- print(f"[build_op_stats] Processed {files_processed} files, {lines_processed} lines → {len(stats)} unique operations", file=sys.stderr)
115
-
116
- return stats
117
-
118
-
119
- def find_cleo_dir(start='.'):
120
- """Walk up from start to find directory containing .cleo/tasks.db."""
121
- p = Path(start).resolve()
122
- while p != p.parent:
123
- if (p / '.cleo' / 'tasks.db').exists():
124
- return p
125
- p = p.parent
126
- return Path(start).resolve()
127
-
128
-
129
- def main():
130
- parser = argparse.ArgumentParser(
131
- description='Aggregate grade run operations.jsonl files into per-operation stats.'
132
- )
133
- parser.add_argument(
134
- '--grade-runs-dir',
135
- default=None,
136
- help='Directory containing grade run subdirectories (default: .cleo/metrics/grade-runs)'
137
- )
138
- parser.add_argument(
139
- '--output',
140
- default=None,
141
- help='Output JSON path (default: .cleo/metrics/per_operation_stats.json)'
142
- )
143
- parser.add_argument(
144
- '--pretty',
145
- action='store_true',
146
- help='Pretty-print JSON output'
147
- )
148
- parser.add_argument(
149
- '--verbose',
150
- action='store_true',
151
- help='Print progress to stderr'
152
- )
153
- args = parser.parse_args()
154
-
155
- workspace = find_cleo_dir('.')
156
-
157
- grade_runs_dir = args.grade_runs_dir or str(workspace / '.cleo' / 'metrics' / 'grade-runs')
158
- output_path = args.output or str(workspace / '.cleo' / 'metrics' / 'per_operation_stats.json')
159
-
160
- stats = compute_stats(grade_runs_dir, verbose=args.verbose)
161
-
162
- indent = 2 if args.pretty else None
163
- output_json = json.dumps(stats, indent=indent)
164
-
165
- out = Path(output_path)
166
- out.parent.mkdir(parents=True, exist_ok=True)
167
- out.write_text(output_json)
168
-
169
- print(f"Wrote {len(stats)} operation stats to {output_path}")
170
- return 0
171
-
172
-
173
- if __name__ == '__main__':
174
- sys.exit(main())
@@ -1,41 +0,0 @@
1
- {
2
- "total_grades": 31,
3
- "score_distribution": {
4
- "F (0)": 7,
5
- "D (45-59)": 10,
6
- "C (60-74)": 5,
7
- "B (75-89)": 8,
8
- "A (90+)": 1
9
- },
10
- "score_stats": {
11
- "mean": 64.6,
12
- "min": 50,
13
- "max": 95,
14
- "grades_with_data": 24,
15
- "zero_score_count": 7
16
- },
17
- "dimension_averages": {
18
- "sessionDiscipline": 5.8,
19
- "discoveryEfficiency": 9.0,
20
- "taskHygiene": 15.4,
21
- "errorProtocol": 15.3,
22
- "disclosureUse": 4.5
23
- },
24
- "flag_frequency": {
25
- "No admin.help calls": 21,
26
- "session.list never called": 18,
27
- "No MCP query calls": 13,
28
- "session.end never called": 12,
29
- "No audit entries": 7,
30
- "tasks.list used (prefer find)": 5,
31
- "Subtasks without exists check": 1,
32
- "Duplicate task creates": 1
33
- },
34
- "avg_audit_entries": 9.5,
35
- "token_estimate": {
36
- "avg_per_session_chars": 0,
37
- "avg_per_session_tokens": 1425.0,
38
- "method": "entry_count * 150 proxy",
39
- "note": "OTEL not enabled; enable with CLAUDE_CODE_ENABLE_TELEMETRY=1 for real counts"
40
- }
41
- }
@@ -1,37 +0,0 @@
1
- # CLEO Grade v2.1 — Comparative Analysis Report
2
-
3
- **Generated:** 2026-03-07 23:47 UTC
4
- **Source:** `/tmp/ct-grade-eval`
5
-
6
- > **DEPRECATED**: This report was generated when MCP was still supported. MCP has been removed.
7
- > All operations now use the CLI exclusively. These results are retained for historical reference only.
8
-
9
- ---
10
-
11
- ## Historical: MCP vs CLI Blind A/B Results
12
-
13
- **Overall winner: MCP**
14
-
15
- | Metric | Value |
16
- |--------|-------|
17
- | Total runs | 3 |
18
- | MCP wins | 3 (100.0%) |
19
- | CLI wins | 0 (0.0%) |
20
- | Ties | 0 |
21
- | Avg token delta (MCP–CLI) | +416.0 tokens |
22
- | Interpretation | MCP uses more tokens on average |
23
-
24
- ### Per-Operation Results
25
-
26
- | Operation | MCP wins | CLI wins | Ties | Token delta | MCP chars | CLI chars | MCP ms | CLI ms |
27
- |-----------|----------|----------|------|-------------|-----------|-----------|--------|--------|
28
- | `admin.version` **MCP** | 3 | 0 | 0 | +416t | 1664 | 0 | 930ms | 786ms |
29
-
30
- ### Recommendations
31
-
32
- - **MCP adds significant token overhead.** Consider whether MCP envelope verbosity can be reduced for high-frequency operations.
33
- - **MCP output quality is consistently higher.** Reinforces MCP-first agent protocol recommendation.
34
-
35
- ---
36
-
37
- *Report generated by ct-grade v2.1 `generate_report.py`*