@cleocode/skills 2026.5.81 → 2026.5.83

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/README.md +0 -1
  2. package/package.json +1 -1
  3. package/profiles/recommended.json +1 -1
  4. package/skills/manifest.json +45 -1
  5. package/skills/ct-grade-v2-1/MIGRATION.md +0 -28
  6. package/skills/ct-grade-v2-1/SKILL.md +0 -235
  7. package/skills/ct-grade-v2-1/agents/analysis-reporter.md +0 -203
  8. package/skills/ct-grade-v2-1/agents/blind-comparator.md +0 -157
  9. package/skills/ct-grade-v2-1/agents/scenario-runner.md +0 -160
  10. package/skills/ct-grade-v2-1/evals/evals.json +0 -74
  11. package/skills/ct-grade-v2-1/grade-viewer/__pycache__/build_op_stats.cpython-314.pyc +0 -0
  12. package/skills/ct-grade-v2-1/grade-viewer/__pycache__/generate_grade_review.cpython-314.pyc +0 -0
  13. package/skills/ct-grade-v2-1/grade-viewer/build_op_stats.py +0 -174
  14. package/skills/ct-grade-v2-1/grade-viewer/eval-analysis.json +0 -41
  15. package/skills/ct-grade-v2-1/grade-viewer/eval-report.md +0 -37
  16. package/skills/ct-grade-v2-1/grade-viewer/generate_grade_review.py +0 -1023
  17. package/skills/ct-grade-v2-1/grade-viewer/generate_grade_viewer.py +0 -548
  18. package/skills/ct-grade-v2-1/grade-viewer/grade-review-eval.html +0 -613
  19. package/skills/ct-grade-v2-1/grade-viewer/grade-review.html +0 -1532
  20. package/skills/ct-grade-v2-1/grade-viewer/viewer.html +0 -620
  21. package/skills/ct-grade-v2-1/manifest-entry.json +0 -31
  22. package/skills/ct-grade-v2-1/references/ab-testing.md +0 -173
  23. package/skills/ct-grade-v2-1/references/domains-ssot.md +0 -156
  24. package/skills/ct-grade-v2-1/references/grade-spec-v2.md +0 -167
  25. package/skills/ct-grade-v2-1/references/playbook-v2.md +0 -325
  26. package/skills/ct-grade-v2-1/references/token-tracking.md +0 -200
  27. package/skills/ct-grade-v2-1/scripts/generate_report.py +0 -419
  28. package/skills/ct-grade-v2-1/scripts/run_ab_test.py +0 -493
  29. package/skills/ct-grade-v2-1/scripts/run_scenario.py +0 -396
  30. package/skills/ct-grade-v2-1/scripts/setup_run.py +0 -207
  31. package/skills/ct-grade-v2-1/scripts/token_tracker.py +0 -175
package/README.md CHANGED
@@ -72,7 +72,6 @@ yarn add @cleocode/skills
72
72
  | Skill | Purpose | Description |
73
73
  |-------|---------|-------------|
74
74
  | **ct-grade** | Grading | Session quality evaluation |
75
- | **ct-grade-v2-1** | Grading V2 | Enhanced grading with scenarios |
76
75
  | **ct-stickynote** | Notes | Quick ephemeral sticky notes |
77
76
 
78
77
  ### Integration Skills
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cleocode/skills",
3
- "version": "2026.5.81",
3
+ "version": "2026.5.83",
4
4
  "description": "CLEO skill definitions - bundled with CLEO monorepo",
5
5
  "main": "index.js",
6
6
  "types": "index.d.ts",
@@ -2,6 +2,6 @@
2
2
  "name": "recommended",
3
3
  "description": "Full LOOM (RCASD-IVTR+C pipeline) skills for epic-driven development",
4
4
  "extends": "core",
5
- "skills": ["ct-epic-architect", "ct-research-agent", "ct-spec-writer", "ct-validator", "loom"],
5
+ "skills": ["ct-council", "ct-epic-architect", "ct-research-agent", "ct-spec-writer", "ct-validator", "loom"],
6
6
  "includeProtocols": ["research", "consensus", "specification", "decomposition", "validation", "adr"]
7
7
  }
@@ -3,7 +3,7 @@
3
3
  "_meta": {
4
4
  "schemaVersion": "2.4.0",
5
5
  "lastUpdated": "2026-04-07",
6
- "totalSkills": 21,
6
+ "totalSkills": 22,
7
7
  "generatedFrom": "T260 — lifecycle pipeline rework: dedicated skills for ADR, IVT loop, consensus, release, artifact-publish, provenance",
8
8
  "architectureNote": "Universal Subagent Architecture: All spawns use provider-neutral delegation with skill/protocol injection. Pipeline stages and cross-cutting protocols each have a dedicated skill — no overloading."
9
9
  },
@@ -700,6 +700,50 @@
700
700
  "requires_session": false,
701
701
  "requires_epic": false
702
702
  }
703
+ },
704
+ {
705
+ "name": "ct-council",
706
+ "version": "1.0.0",
707
+ "description": "Convene \"The Council\" — a 5-advisor, shuffled gate-based peer-review, chairman-synthesis workflow for reviewing a plan, decision, architecture, or piece of work inside the current project. Operates on the current codebase — each advisor grounds their analysis in actual files/commits before opining. Output is validated by scripts/validate.py.",
708
+ "path": "skills/ct-council",
709
+ "tags": ["council", "review", "peer-review", "multi-perspective", "stress-test"],
710
+ "status": "active",
711
+ "tier": 2,
712
+ "token_budget": 8000,
713
+ "references": [
714
+ "skills/ct-council/references/chairman.md",
715
+ "skills/ct-council/references/contrarian.md",
716
+ "skills/ct-council/references/evidence-pack.md",
717
+ "skills/ct-council/references/examples.md",
718
+ "skills/ct-council/references/executor.md",
719
+ "skills/ct-council/references/expansionist.md",
720
+ "skills/ct-council/references/first-principles.md",
721
+ "skills/ct-council/references/outsider.md",
722
+ "skills/ct-council/references/peer-review.md"
723
+ ],
724
+ "capabilities": {
725
+ "inputs": ["proposal", "plan", "architecture-decision", "task-id"],
726
+ "outputs": ["council-verdict", "advisor-reports", "convergence-report"],
727
+ "dependencies": [],
728
+ "dispatch_triggers": [
729
+ "convene the council",
730
+ "council review",
731
+ "run the five advisors",
732
+ "stress-test this",
733
+ "get multiple perspectives"
734
+ ],
735
+ "compatible_subagent_types": ["general-purpose"],
736
+ "chains_to": [],
737
+ "dispatch_keywords": {
738
+ "primary": ["council", "advisors", "stress-test", "peer-review"],
739
+ "secondary": ["contrarian", "first-principles", "expansionist", "outsider", "executor", "chairman"]
740
+ }
741
+ },
742
+ "constraints": {
743
+ "max_context_tokens": 80000,
744
+ "requires_session": false,
745
+ "requires_epic": false
746
+ }
703
747
  }
704
748
  ]
705
749
  }
@@ -1,28 +0,0 @@
1
- ---
2
- # MIGRATION NOTICE — ct-grade-v2-1
3
-
4
- This directory is a **decommissioned staging copy** of `ct-grade` v2.1.
5
-
6
- ## Status
7
-
8
- **Superseded.** All content has been merged into `packages/skills/skills/ct-grade/`.
9
-
10
- ## What Changed (T429 skill dedupe)
11
-
12
- - `ct-grade-v2-1/SKILL.md` description, `argument-hint`, `allowed-tools`, and version
13
- (2.1.0) were promoted into `ct-grade/SKILL.md`.
14
- - `ct-grade-v2-1/manifest-entry.json` remains here as an archived snapshot; the
15
- canonical manifest entry lives in `packages/skills/skills/manifest.json` under name
16
- `ct-grade`.
17
- - The `grade-viewer/` tooling in this directory was already reachable from `ct-grade/`
18
- via its `agents/` and `evals/` directories. No content was lost.
19
-
20
- ## Migration Date
21
-
22
- 2026-04-08 — T429 hygiene wave, epic T382.
23
-
24
- ## Action Required
25
-
26
- None. Do NOT load this skill. Use `ct-grade` instead.
27
- If you need the A/B or blind-compare modes documented here, they are now part of
28
- `ct-grade`'s SKILL.md description and invocation modes.
@@ -1,235 +0,0 @@
1
- ---
2
- name: ct-grade
3
- description: >-
4
- CLEO session grading and A/B behavioral analysis with token tracking. Evaluates agent
5
- session quality via a 5-dimension rubric (S1 session discipline, S2 discovery efficiency,
6
- S3 task hygiene, S4 error protocol, S5 progressive disclosure). Supports three modes:
7
- (1) scenario — run playbook scenarios S1-S5 via CLI; (2) ab — blind A/B
8
- comparison of different CLI configurations for same domain operations with token cost
9
- measurement; (3) blind — spawn two agents with different configurations, blind-comparator
10
- picks winner, analyzer produces recommendation. Use when grading agent sessions, running
11
- grade playbook scenarios, comparing behavioral differences, measuring token
12
- usage across configurations, or performing multi-run blind A/B evaluation with statistical
13
- analysis and comparative report. Triggers on: grade session, evaluate agent behavior,
14
- A/B test CLEO configurations, run grade scenario, token usage analysis, behavioral rubric,
15
- protocol compliance scoring.
16
- argument-hint: "[mode=scenario|ab|blind] [scenario=s1-s5|all] [runs=N] [session-id=<id>]"
17
- allowed-tools: ["Bash(python *)", "Bash(cleo-dev *)", "Bash(cleo *)", "Bash(kill *)", "Bash(lsof *)", "Agent", "Read", "Write", "Glob"]
18
- ---
19
-
20
- # ct-grade v2.1 — CLEO Grading and A/B Testing
21
-
22
- Session grading and A/B behavioral analysis for CLEO protocol compliance. Three operating modes cover everything from single-session scoring to multi-run blind comparisons between different CLI configurations.
23
-
24
- ## On Every /ct-grade Invocation
25
-
26
- Before parsing arguments, start the grade viewer server:
27
-
28
- ```bash
29
- # Kill any existing viewer on port 3119
30
- lsof -ti :3119 | xargs kill -TERM 2>/dev/null || true
31
-
32
- # Start grade viewer in background
33
- python $CLAUDE_SKILL_DIR/grade-viewer/generate_grade_review.py . \
34
- --port 3119 --no-browser &
35
- echo "Grade viewer: http://localhost:3119"
36
- ```
37
-
38
- When user says "end grading", "stop", "done", or "close viewer":
39
- ```bash
40
- lsof -ti :3119 | xargs kill -TERM 2>/dev/null || true
41
- echo "Grade viewer stopped."
42
- ```
43
-
44
- ---
45
-
46
- ## Operating Modes
47
-
48
- | Mode | Purpose | Key Output |
49
- |---|---|---|
50
- | `scenario` | Run playbook scenarios S1-S5 as graded sessions | GradeResult per scenario |
51
- | `ab` | Run same domain operations with two configurations, compare | comparison.json + token delta |
52
- | `blind` | Two agents run same task, blind comparator picks winner | analysis.json + winner |
53
-
54
- ## Parameters
55
-
56
- | Parameter | Values | Default | Description |
57
- |---|---|---|---|
58
- | `mode` | `scenario\|ab\|blind` | `scenario` | Operating mode |
59
- | `scenario` | `s1\|s2\|s3\|s4\|s5\|all` | `all` | Grade playbook scenario(s) to run |
60
- | `interface` | `cli` | `cli` | Interface to exercise (CLI only) |
61
- | `domains` | comma list | `tasks,session` | Domains to test in `ab` mode |
62
- | `runs` | integer | `3` | Runs per configuration for statistical confidence |
63
- | `session-id` | string | — | Grade a specific existing session (skips execution) |
64
- | `output-dir` | path | `ab_results/<ts>` | Where to write all run artifacts |
65
-
66
- ## Quick Start
67
-
68
- **Grade an existing session:**
69
- ```
70
- /ct-grade session-id=<id>
71
- ```
72
-
73
- **Run scenario S4 (Full Lifecycle):**
74
- ```
75
- /ct-grade mode=scenario scenario=s4
76
- ```
77
-
78
- **A/B compare two configurations for tasks + session domains (3 runs each):**
79
- ```
80
- /ct-grade mode=ab domains=tasks,session runs=3
81
- ```
82
-
83
- **Full blind A/B test across all scenarios:**
84
- ```
85
- /ct-grade mode=blind scenario=all runs=3
86
- ```
87
-
88
- ---
89
-
90
- ## Execution Flow
91
-
92
- ### Mode: scenario
93
-
94
- 1. Set up output dir with `python $CLAUDE_SKILL_DIR/scripts/setup_run.py --mode scenario --scenario <id> --output-dir <dir>`
95
- 2. For each scenario, spawn a `scenario-runner` agent:
96
- - Agent start: `cleo session start --scope global --name "<scenario-id>" --grade`
97
- - Agent executes the scenario operations (see [references/playbook-v2.md](references/playbook-v2.md))
98
- - Agent end: `cleo session end`
99
- - Agent runs: `ct grade <sessionId>`
100
- - Agent saves: `GradeResult` to `<output-dir>/<scenario>/grade.json`
101
- 3. Capture `total_tokens` + `duration_ms` from task notification → `timing.json`
102
- 4. Run: `python $CLAUDE_SKILL_DIR/scripts/generate_report.py --run-dir <dir> --mode scenario`
103
-
104
- ### Mode: ab
105
-
106
- 1. Set up run dir with `python $CLAUDE_SKILL_DIR/scripts/setup_run.py --mode ab --output-dir <dir>`
107
- 2. For each target domain, spawn TWO agents in the SAME turn:
108
- - **Arm A**: `agents/scenario-runner.md` with configuration A
109
- - **Arm B**: `agents/scenario-runner.md` with configuration B
110
- - Capture tokens from both task notifications immediately
111
- 3. Pass both outputs to `agents/blind-comparator.md` (does NOT know which configuration is which)
112
- 4. Comparator writes `comparison.json`
113
- 5. Run `python $CLAUDE_SKILL_DIR/scripts/generate_report.py --run-dir <dir> --mode ab`
114
-
115
- ### Mode: blind
116
-
117
- Same as `ab` but configurations may differ (e.g., different session scopes, different agent prompts). The comparator is always blind to configuration identity.
118
-
119
- ---
120
-
121
- ## Token Capture — MANDATORY
122
-
123
- After EVERY Agent task notification, immediately update `timing.json`:
124
-
125
- ```python
126
- timing = {
127
- "total_tokens": task.total_tokens, # from task notification — EPHEMERAL
128
- "duration_ms": task.duration_ms, # from task notification
129
- "arm": "arm-A",
130
- "interface": "cli",
131
- "scenario": "s4",
132
- "run": 1,
133
- "executor_start": start_iso,
134
- "executor_end": end_iso,
135
- }
136
- # Write to: <output-dir>/<scenario>/arm-<interface>/timing.json
137
- ```
138
-
139
- **`total_tokens` is EPHEMERAL** — it cannot be recovered if missed. Capture it immediately.
140
-
141
- If running without task notifications (no total_tokens available):
142
- - Fall back: `output_chars / 3.5` from operations.jsonl (JSON responses)
143
- - Record `"method": "output_chars_estimate"` in timing.json
144
-
145
- ---
146
-
147
- ## Grade Rubric Summary
148
-
149
- 5 dimensions × 20 pts = 100 max. See [references/grade-spec-v2.md](references/grade-spec-v2.md) for full scoring logic.
150
-
151
- | Dim | Points | What it measures |
152
- |---|---|---|
153
- | S1 Session Discipline | 20 | `session.list` before task ops (+10), `session.end` present (+10) |
154
- | S2 Discovery Efficiency | 20 | `find:list` ratio ≥80% (+15), `tasks.show` used (+5) |
155
- | S3 Task Hygiene | 20 | Starts 20, -5 per add without description, -3 if subtask no exists check |
156
- | S4 Error Protocol | 20 | Starts 20, -5 per unrecovered E_NOT_FOUND, -5 if duplicates |
157
- | S5 Progressive Disclosure | 20 | `admin.help`/skill lookup (+10), progressive disclosure used (+10) |
158
-
159
- **Grade letters:** A>=90, B>=75, C>=60, D>=45, F<45
160
-
161
- ---
162
-
163
- ## Output Structure
164
-
165
- ```
166
- <output-dir>/
167
- run-manifest.json # run config, arms, timing summary
168
- report.md # human-readable comparative report
169
- token-summary.json # aggregated token stats across all runs
170
- <scenario-or-domain>/
171
- arm-A/
172
- grade.json # GradeResult (from check.grade)
173
- timing.json # token + duration data
174
- operations.jsonl # operations executed (one per line)
175
- arm-B/
176
- grade.json
177
- timing.json
178
- operations.jsonl
179
- comparison.json # blind comparator output
180
- analysis.json # analyzer output
181
- ```
182
-
183
- ---
184
-
185
- ## Agents
186
-
187
- | Agent | Role | Input | Output |
188
- |---|---|---|---|
189
- | [agents/scenario-runner.md](agents/scenario-runner.md) | Executes grade scenario | scenario, interface | grade.json, timing.json |
190
- | [agents/blind-comparator.md](agents/blind-comparator.md) | Blind A/B judge | outputs A and B | comparison.json |
191
- | [agents/analysis-reporter.md](agents/analysis-reporter.md) | Post-hoc synthesis | all comparison.json | analysis.json |
192
-
193
- ---
194
-
195
- ## Scripts
196
-
197
- ```bash
198
- # Set up run directory and print execution plan
199
- python $CLAUDE_SKILL_DIR/scripts/setup_run.py --mode <mode> --scenario <s> --output-dir <dir>
200
-
201
- # Aggregate token data after runs complete
202
- python $CLAUDE_SKILL_DIR/scripts/token_tracker.py --run-dir <dir>
203
-
204
- # Generate final report (markdown)
205
- python $CLAUDE_SKILL_DIR/scripts/generate_report.py --run-dir <dir> --mode <mode>
206
- ```
207
-
208
- ---
209
-
210
- ## Viewers
211
-
212
- ### Grade Results Viewer (A/B run artifacts) — port 3119
213
- ```bash
214
- python $CLAUDE_SKILL_DIR/grade-viewer/generate_grade_viewer.py --run-dir <ab-run-dir>
215
- python $CLAUDE_SKILL_DIR/grade-viewer/generate_grade_viewer.py --run-dir <ab-run-dir> --static results.html
216
- ```
217
- Shows per-scenario grade cards with dimension bars, A/B comparison tables, token economy stats, blind comparator results, and recommendations. Refreshes on browser reload.
218
-
219
- ### General Grade Review (GRADES.jsonl browsing) — port 3119
220
- ```bash
221
- python $CLAUDE_SKILL_DIR/grade-viewer/generate_grade_review.py <workspace>
222
- python $CLAUDE_SKILL_DIR/grade-viewer/generate_grade_review.py <workspace> --static grade-report.html
223
- ```
224
- Shows historical grades from GRADES.jsonl, A/B summaries from any workspace subdirectory.
225
-
226
- ---
227
-
228
- ## CLI Grade Operations
229
-
230
- | Command | Description |
231
- |---------|-------------|
232
- | `ct grade <sessionId>` | Grade a specific session |
233
- | `ct grade --list` | List past grade results |
234
- | `ct session start --scope global --name "<n>" --grade` | Start graded session |
235
- | `ct session end` | End session |
@@ -1,203 +0,0 @@
1
- # Analysis Reporter Agent
2
-
3
- You are a post-hoc analyzer for CLEO A/B evaluation results. You synthesize all comparison.json and grade.json files from a completed run into a final `analysis.json` and `report.md`.
4
-
5
- ## Inputs
6
-
7
- - `RUN_DIR`: Path to the completed run directory
8
- - `MODE`: `scenario|ab|blind`
9
- - `OUTPUT_PATH`: Where to write analysis.json (default: `<RUN_DIR>/analysis.json`)
10
- - `REPORT_PATH`: Where to write report.md (default: `<RUN_DIR>/report.md`)
11
-
12
- ## What You Read
13
-
14
- From `<RUN_DIR>`:
15
- ```
16
- run-manifest.json
17
- token-summary.json (from token_tracker.py)
18
- <scenario-or-domain>/
19
- arm-A/grade.json
20
- arm-A/timing.json
21
- arm-A/operations.jsonl
22
- arm-B/grade.json
23
- arm-B/timing.json
24
- arm-B/operations.jsonl
25
- comparison.json
26
- ```
27
-
28
- ## Analysis Process
29
-
30
- ### 1. Aggregate grade results
31
-
32
- For each scenario/domain, collect:
33
- - A's total_score and per-dimension scores
34
- - B's total_score and per-dimension scores
35
- - comparison winner
36
- - Token counts for each arm
37
-
38
- ### 2. Compute cross-run statistics
39
-
40
- If multiple runs exist:
41
- - mean, stddev, min, max for total_score per arm
42
- - mean, stddev for total_tokens per arm
43
- - Win rate for each arm across runs
44
-
45
- ### 3. Identify patterns
46
-
47
- Look for:
48
- - Dimensions where one arm consistently outperforms
49
- - Scenarios where MCP and CLI diverge most
50
- - Operations that appear in failures but not successes
51
- - Token efficiency: score-per-token comparison
52
-
53
- ### 4. Generate recommendations
54
-
55
- Based on patterns:
56
- - Which interface (MCP/CLI) performs better overall?
57
- - Which dimensions need protocol improvement?
58
- - Which scenarios expose the most variance?
59
- - What specific anti-patterns appear most?
60
-
61
- ## Output: analysis.json
62
-
63
- ```json
64
- {
65
- "run_summary": {
66
- "mode": "ab",
67
- "scenarios_run": ["s1", "s4"],
68
- "total_runs": 6,
69
- "arms": {
70
- "A": {"label": "MCP interface", "runs": 3},
71
- "B": {"label": "CLI interface", "runs": 3}
72
- }
73
- },
74
- "grade_statistics": {
75
- "A": {
76
- "total_score": {"mean": 88.3, "stddev": 4.5, "min": 83, "max": 93},
77
- "dimensions": {
78
- "sessionDiscipline": {"mean": 18.3, "stddev": 2.3},
79
- "discoveryEfficiency": {"mean": 18.0, "stddev": 1.5},
80
- "taskHygiene": {"mean": 18.7, "stddev": 2.1},
81
- "errorProtocol": {"mean": 18.7, "stddev": 2.3},
82
- "disclosureUse": {"mean": 14.7, "stddev": 4.5}
83
- }
84
- },
85
- "B": {
86
- "total_score": {"mean": 71.7, "stddev": 8.1, "min": 62, "max": 80},
87
- "dimensions": {
88
- "sessionDiscipline": {"mean": 14.0, "stddev": 5.3},
89
- "discoveryEfficiency": {"mean": 17.3, "stddev": 2.1},
90
- "taskHygiene": {"mean": 18.0, "stddev": 2.0},
91
- "errorProtocol": {"mean": 16.7, "stddev": 3.8},
92
- "disclosureUse": {"mean": 5.7, "stddev": 4.7}
93
- }
94
- }
95
- },
96
- "token_statistics": {
97
- "A": {"mean": 4200, "stddev": 380, "min": 3800, "max": 4600},
98
- "B": {"mean": 2900, "stddev": 220, "min": 2650, "max": 3100},
99
- "delta": {"mean": 1300, "percent": "+44.8%"},
100
- "score_per_1k_tokens": {"A": 21.0, "B": 24.7}
101
- },
102
- "win_rates": {
103
- "A_wins": 5,
104
- "B_wins": 1,
105
- "ties": 0,
106
- "A_win_rate": 0.833
107
- },
108
- "dimension_analysis": [
109
- {
110
- "dimension": "disclosureUse",
111
- "insight": "S5 shows highest variance between arms. MCP arm uses admin.help consistently; CLI arm often skips it.",
112
- "A_mean": 14.7,
113
- "B_mean": 5.7,
114
- "delta": 9.0
115
- },
116
- {
117
- "dimension": "sessionDiscipline",
118
- "insight": "CLI arm frequently calls session.list after task ops, violating S1 ordering.",
119
- "A_mean": 18.3,
120
- "B_mean": 14.0,
121
- "delta": 4.3
122
- }
123
- ],
124
- "pattern_analysis": {
125
- "winner_execution_pattern": "Start session -> session.list -> admin.help -> tasks.find -> tasks.show -> work -> session.end",
126
- "loser_execution_pattern": "Start session -> tasks.find (skip session.list) -> work -> session.end (skip admin.help)",
127
- "common_failures": [
128
- "session.list called after first task op (violates S1 +10)",
129
- "admin.help not called (violates S5 +10)",
130
- "tasks.list used instead of tasks.find (reduces S2)"
131
- ]
132
- },
133
- "improvement_suggestions": [
134
- {
135
- "priority": "high",
136
- "dimension": "S1",
137
- "suggestion": "CLI interface does not prompt for session.list before task ops. Add a pre-task-op reminder.",
138
- "expected_impact": "Would recover +10 S1 points consistently in CLI arm"
139
- },
140
- {
141
- "priority": "high",
142
- "dimension": "S5",
143
- "suggestion": "CLI arm never calls admin.help. Skill should explicitly prompt 'call admin.help at session start'.",
144
- "expected_impact": "Would recover +10 S5 points"
145
- },
146
- {
147
- "priority": "medium",
148
- "dimension": "token_efficiency",
149
- "suggestion": "MCP arm uses +44.8% more tokens but scores +16.6 points higher. Net score-per-token still favors MCP for protocol-critical work.",
150
- "expected_impact": "Context for choosing interface based on task priority"
151
- }
152
- ]
153
- }
154
- ```
155
-
156
- ## Output: report.md
157
-
158
- Write a human-readable comparative report with:
159
-
160
- 1. **Executive Summary** — winner, score delta, token delta
161
- 2. **Per-Scenario Results** — table of A vs B scores per scenario
162
- 3. **Dimension Breakdown** — where each arm excels/fails
163
- 4. **Token Economy** — total_tokens comparison, score-per-token
164
- 5. **Pattern Analysis** — common success/failure patterns
165
- 6. **Recommendations** — actionable improvements ranked by impact
166
-
167
- Use this structure:
168
-
169
- ```markdown
170
- # CLEO Grade A/B Analysis Report
171
- **Run**: <timestamp> **Mode**: <mode> **Scenarios**: <list>
172
-
173
- ## Executive Summary
174
- | Metric | Arm A (MCP) | Arm B (CLI) | Delta |
175
- |---|---|---|---|
176
- | Mean Score | 88.3/100 | 71.7/100 | +16.6 |
177
- | Grade | A | C | — |
178
- | Mean Tokens | 4,200 | 2,900 | +1,300 (+44.8%) |
179
- | Score/1k tokens | 21.0 | 24.7 | -3.7 |
180
- | Win Rate | 83.3% | 16.7% | — |
181
-
182
- **Winner: Arm A (MCP)** — Higher protocol adherence in 5/6 runs.
183
- Token cost is higher but justified by significant score improvement.
184
-
185
- ## Per-Scenario Results
186
- ...
187
-
188
- ## Dimension Analysis
189
- ...
190
-
191
- ## Recommendations
192
- ...
193
- ```
194
-
195
- After writing both files, output:
196
- ```
197
- ANALYSIS: <analysis.json path>
198
- REPORT: <report.md path>
199
- WINNER_ARM: <A|B|tie>
200
- WINNER_CONFIG: <mcp|cli|other>
201
- MEAN_DELTA: <+N points>
202
- TOKEN_DELTA: <+N tokens>
203
- ```