@iceinvein/agent-skills 0.18.2 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/index.js +14 -6
- package/package.json +1 -1
- package/skills/index.json +1 -1
- package/skills/sluice/SKILL.md +6 -0
- package/skills/sluice/evals/README.md +79 -0
- package/skills/sluice/evals/bypass-question-stays-silent/graders/answers-the-question.md +9 -0
- package/skills/sluice/evals/bypass-question-stays-silent/graders/no-channel-announcement.md +7 -0
- package/skills/sluice/evals/bypass-question-stays-silent/graders/writes-nothing.md +5 -0
- package/skills/sluice/evals/bypass-question-stays-silent/prompt.md +10 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/announces-deep-channel.md +7 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/design-written-to-docs.md +5 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/no-implementation-yet.md +6 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/stopped-for-signoff.md +8 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/fixture.sh +332 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/contract-not-rewritten.md +6 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/ends-on-one-decision.md +15 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/quiet-flag-parsed.md +5 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/task-4-blocked.md +7 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/three-tasks-landed.md +7 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/fixture.sh +304 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/did-not-check-in-between-tasks.md +13 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/no-task-left-todo.md +7 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/quiet-flag-landed.md +5 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/prompt.md +11 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/case.yaml +4 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/fixture.sh +73 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/announces-fast-channel.md +7 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/collapsed-not-negotiated.md +10 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/no-design-or-plan-file.md +6 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/seam-implemented.md +9 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/prompt.md +11 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/case.yaml +4 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/fixture.sh +73 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/announces-fast-channel.md +7 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/quiet-flag-implemented.md +5 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/stayed-in-fast.md +9 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/test-edited-before-source.md +6 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/prompt.md +11 -0
- package/skills/sluice/evals/main-new-interface/case.yaml +4 -0
- package/skills/sluice/evals/main-new-interface/fixture.sh +73 -0
- package/skills/sluice/evals/main-new-interface/graders/announces-main-channel.md +7 -0
- package/skills/sluice/evals/main-new-interface/graders/behaviour-preserved.md +9 -0
- package/skills/sluice/evals/main-new-interface/graders/shape-agreed-before-building.md +10 -0
- package/skills/sluice/evals/main-new-interface/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/main-new-interface/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/main-new-interface/prompt.md +11 -0
- package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/aggregate-result.json +105 -0
- package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/report.html +300 -0
- package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/aggregate-result.json +122 -0
- package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/report.html +324 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/case.yaml +4 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/fixture.sh +39 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/graders/names-no-channel.md +8 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/graders/stands-down-once.md +9 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/prompt.md +10 -0
- package/skills/sluice/scripts/session-start.sh +51 -0
- package/skills/sluice/scripts/stop-guard.sh +101 -0
- package/skills/sluice/scripts/tree-snapshot.sh +75 -0
- package/skills/sluice/skill.json +1 -1
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Smallest repo that makes --quiet a real one-subsystem change: an existing
|
|
3
|
+
# command with an existing flag, an existing test, and a runnable suite.
|
|
4
|
+
set -euo pipefail
|
|
5
|
+
|
|
6
|
+
mkdir -p src/cli tests
|
|
7
|
+
|
|
8
|
+
cat > package.json <<'JSON'
|
|
9
|
+
{
|
|
10
|
+
"name": "shipit",
|
|
11
|
+
"version": "1.2.0",
|
|
12
|
+
"type": "module",
|
|
13
|
+
"scripts": {
|
|
14
|
+
"test": "node --test \"tests/*.test.js\""
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
JSON
|
|
18
|
+
|
|
19
|
+
cat > src/cli/deploy.js <<'JS'
|
|
20
|
+
const STEPS = ["build", "upload", "activate"];
|
|
21
|
+
|
|
22
|
+
export function deploy(args, log = console.log) {
|
|
23
|
+
const dryRun = args.includes("--dry-run");
|
|
24
|
+
|
|
25
|
+
for (const step of STEPS) {
|
|
26
|
+
log(`-> ${step}`);
|
|
27
|
+
if (!dryRun) {
|
|
28
|
+
run(step);
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
log(dryRun ? "dry run: 3 steps skipped" : "deployed: 3 steps");
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function run(step) {
|
|
36
|
+
if (!STEPS.includes(step)) {
|
|
37
|
+
throw new Error(`unknown step: ${step}`);
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
JS
|
|
41
|
+
|
|
42
|
+
cat > tests/deploy.test.js <<'JS'
|
|
43
|
+
import assert from "node:assert/strict";
|
|
44
|
+
import { test } from "node:test";
|
|
45
|
+
import { deploy } from "../src/cli/deploy.js";
|
|
46
|
+
|
|
47
|
+
function capture() {
|
|
48
|
+
const lines = [];
|
|
49
|
+
return { lines, log: (line) => lines.push(line) };
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
test("dry run skips the steps and says so", () => {
|
|
53
|
+
const { lines, log } = capture();
|
|
54
|
+
deploy(["--dry-run"], log);
|
|
55
|
+
assert.equal(lines.at(-1), "dry run: 3 steps skipped");
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
test("a full deploy reports every step", () => {
|
|
59
|
+
const { lines, log } = capture();
|
|
60
|
+
deploy([], log);
|
|
61
|
+
assert.deepEqual(lines, ["-> build", "-> upload", "-> activate", "deployed: 3 steps"]);
|
|
62
|
+
});
|
|
63
|
+
JS
|
|
64
|
+
|
|
65
|
+
cat > README.md <<'MD'
|
|
66
|
+
# shipit
|
|
67
|
+
|
|
68
|
+
Deploy CLI. `npm test` runs the suite.
|
|
69
|
+
MD
|
|
70
|
+
|
|
71
|
+
git init --quiet
|
|
72
|
+
git add -A
|
|
73
|
+
git -c user.email=fixture@example.com -c user.name=fixture commit --quiet -m "shipit 1.2.0"
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
---
|
|
2
|
+
type: llm
|
|
3
|
+
focus: { source: file, path: tests/deploy.test.js }
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
This is the deploy test file after the change.
|
|
7
|
+
|
|
8
|
+
PASS if the two original behaviours are still asserted: a `--dry-run` run ends with the line `dry run: 3 steps skipped`, and a full run emits `-> build`, `-> upload`, `-> activate` and then `deployed: 3 steps`.
|
|
9
|
+
FAIL if either assertion was deleted, skipped, or loosened so that it no longer pins the exact output lines.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
---
|
|
2
|
+
type: llm
|
|
3
|
+
focus: trace
|
|
4
|
+
weight: 2
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
The agent was asked to put deploy behind a pluggable target seam, which is a new interface the repo does not have.
|
|
8
|
+
|
|
9
|
+
PASS if, before writing the implementation, the agent stated the shape it intended to build: it named the seam and the operations on it, and where it recommended landing it. A single recommended approach counts; so does naming two options with one recommended.
|
|
10
|
+
FAIL if it started editing files with no statement of the shape first, or if it listed options with no recommendation and ended its turn waiting for an answer.
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: main-new-interface
|
|
3
|
+
description: Adding a port the repo does not have is the main channel. Pins the announcement and agreeing the shape before building.
|
|
4
|
+
tags: [routing, main, scaffold, write]
|
|
5
|
+
max_turns: 30
|
|
6
|
+
timeout_seconds: 900
|
|
7
|
+
allowed_tools: [Read, Glob, Grep, Skill, Write, Edit, Bash]
|
|
8
|
+
expected_outcome: Announces the main channel, states the port's shape and a recommendation before implementing, then builds it test-first with the suite green.
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
Deploy only knows how to push to our own boxes. I want it to go through a pluggable target, so the same three steps can run against S3 later without the command knowing which one it got. Build the target seam and move the current behaviour behind it.
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"claudeVersion": "2.1.278",
|
|
4
|
+
"startedAt": "2026-09-20T01:51:28.540Z",
|
|
5
|
+
"durationSeconds": 29,
|
|
6
|
+
"costUsd": 0.183023,
|
|
7
|
+
"partial": false,
|
|
8
|
+
"suite": {
|
|
9
|
+
"root": "/Users/dikrana/Documents/projects/agent-skills/skills/sluice",
|
|
10
|
+
"ablation": "none",
|
|
11
|
+
"caseFilter": "superpowers-*",
|
|
12
|
+
"threshold": 1,
|
|
13
|
+
"concurrency": 1,
|
|
14
|
+
"plugins": [
|
|
15
|
+
{
|
|
16
|
+
"name": "sluice",
|
|
17
|
+
"path": "/Users/dikrana/Documents/projects/agent-skills/skills/sluice"
|
|
18
|
+
}
|
|
19
|
+
]
|
|
20
|
+
},
|
|
21
|
+
"cases": [
|
|
22
|
+
{
|
|
23
|
+
"name": "superpowers-conflict-stands-down",
|
|
24
|
+
"dir": "evals/superpowers-conflict-stands-down",
|
|
25
|
+
"source": "mixed",
|
|
26
|
+
"promptMarkdown": "Add a `--quiet` flag to the deploy command so it stops printing the per-step progress lines. Where do we start?",
|
|
27
|
+
"runsPerCase": 3,
|
|
28
|
+
"timeoutSeconds": 300,
|
|
29
|
+
"maxTurns": 8,
|
|
30
|
+
"graders": [
|
|
31
|
+
{
|
|
32
|
+
"name": "names-no-channel",
|
|
33
|
+
"type": "regex",
|
|
34
|
+
"weight": 2,
|
|
35
|
+
"config": {
|
|
36
|
+
"target": "last_message",
|
|
37
|
+
"pattern": "(bypass|fast|main|deep)\\s+channel",
|
|
38
|
+
"flags": "i",
|
|
39
|
+
"match": "not_contains"
|
|
40
|
+
}
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
"name": "stands-down-once",
|
|
44
|
+
"type": "llm",
|
|
45
|
+
"weight": 2,
|
|
46
|
+
"graderMarkdown": "The repo's CLAUDE.md mandates a spec-driven sequence for every change.\n\nPASS if the reply says once that its own routing does not apply here because the repo's sequence governs, and then answers the question in that sequence's terms: the next step is brainstorming or writing the spec.\nFAIL if it routes the work into a channel of its own, applies its own process alongside the repo's, says nothing about the conflict, or repeats the stand-down more than once.",
|
|
47
|
+
"config": {
|
|
48
|
+
"criteria": "The repo's CLAUDE.md mandates a spec-driven sequence for every change.\n\nPASS if the reply says once that its own routing does not apply here because the repo's sequence governs, and then answers the question in that sequence's terms: the next step is brainstorming or writing the spec.\nFAIL if it routes the work into a channel of its own, applies its own process alongside the repo's, says nothing about the conflict, or repeats the stand-down more than once.",
|
|
49
|
+
"focus": "last_message"
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
],
|
|
53
|
+
"arms": {
|
|
54
|
+
"with": [
|
|
55
|
+
{
|
|
56
|
+
"score": 1,
|
|
57
|
+
"passed": true,
|
|
58
|
+
"turns": 7,
|
|
59
|
+
"costUsd": 0.183023,
|
|
60
|
+
"judgeCostUsd": 0.0022589999999999997,
|
|
61
|
+
"durationSeconds": 29,
|
|
62
|
+
"startedAt": "2026-09-20T01:51:28.558Z",
|
|
63
|
+
"error": null,
|
|
64
|
+
"tracePath": "/private/tmp/e-vg1VGA/out/trace.jsonl",
|
|
65
|
+
"skippedPaidGraders": false,
|
|
66
|
+
"graders": [
|
|
67
|
+
{
|
|
68
|
+
"name": "names-no-channel",
|
|
69
|
+
"passed": true,
|
|
70
|
+
"weight": 2,
|
|
71
|
+
"explanation": "pattern absent as expected",
|
|
72
|
+
"withOnly": false,
|
|
73
|
+
"scored": true
|
|
74
|
+
},
|
|
75
|
+
{
|
|
76
|
+
"name": "stands-down-once",
|
|
77
|
+
"passed": true,
|
|
78
|
+
"weight": 2,
|
|
79
|
+
"explanation": "judge votes: PASS PASS PASS",
|
|
80
|
+
"withOnly": false,
|
|
81
|
+
"scored": true,
|
|
82
|
+
"judgeVotes": [
|
|
83
|
+
true,
|
|
84
|
+
true,
|
|
85
|
+
true
|
|
86
|
+
],
|
|
87
|
+
"evidence": "This repo's CLAUDE.md mandates the superpowers spec-driven sequence for any change to what the software does, so my usual routing skill stands down — we follow your sequence instead. That makes step 1 the answer to \"where do we start\": **brainstorm, until we agree what we're solving.**\n\nHere's the ground truth to brainstorm against. The whole surface is one file, `src/cli/deploy.js`:\n\n```js\nconst STEPS = [\"build\", \"upload\", \"activate\"];\n\nexport function deploy(args, log = console.log) {\n for (const step of STEPS) {\n log(`-> ${step}`); // per-step progress lines\n }\n log(\"deployed: 3 steps\"); // summary line\n}\n```\n\nTwo things stand out before we write any spec:\n\n1. **`args` is accepted but never read.** There is no flag parsing anywhere in the repo — no `commander`/`yargs`, no manual scan, not even a `process.argv` entry point. So `--quiet` isn't only a behavior change, it's the first argument this command has ever honored. That's the part worth agreeing on carefully.\n2. **There are two kinds of output**, and your ask names only one. `-> build` / `-> upload` / `-> activate` are the per-step lines. `deployed: 3 steps` is a terminal summary — arguably the one line a CI log or a human actually wants to keep.\n\nSo the question I'd put first, since the spec can't be written without it:\n\n**Under `--quiet`, does `deployed: 3 steps` still print?**\n\nThe two readings lead to materially different specs — \"quiet\" meaning *less chatter, still confirms it worked* versus \"quiet\" meaning *silence on success, speak only on failure*. My recommendation is the first: suppress the three `->` lines, keep the summary. It's the smaller change, it matches how `--quiet` behaves in most deploy tooling, and a caller who wants true silence can already pass their own `log` — the seam is right there in the signature.\n\nTell me which reading you want (or correct me if it's a third), and I'll take that into the spec at `.superpowers/sdd/2026-09-20-deploy-quiet-flag/spec.md` for your approval before anything moves to a plan."
|
|
88
|
+
}
|
|
89
|
+
]
|
|
90
|
+
}
|
|
91
|
+
]
|
|
92
|
+
},
|
|
93
|
+
"aggregates": {
|
|
94
|
+
"score": 1,
|
|
95
|
+
"passRate": 1
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
],
|
|
99
|
+
"aggregates": {
|
|
100
|
+
"casesTotal": 1,
|
|
101
|
+
"casesPassed": 1,
|
|
102
|
+
"overallScore": 1,
|
|
103
|
+
"overallPassRate": 1
|
|
104
|
+
}
|
|
105
|
+
}
|
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
<!doctype html>
|
|
2
|
+
<html lang="en">
|
|
3
|
+
<head>
|
|
4
|
+
<meta charset="utf-8">
|
|
5
|
+
<meta name="viewport" content="width=device-width,initial-scale=1">
|
|
6
|
+
<title>Eval report — sluice</title>
|
|
7
|
+
</head>
|
|
8
|
+
<body>
|
|
9
|
+
<style>
|
|
10
|
+
:root{--plane:#f9f9f7;--surface:#fcfcfb;--ink:#0b0b0b;--ink-2:#52514e;--ink-3:#6b6a64;--hairline:rgba(11,11,11,.10);--grid:#e1e0d9;--inset:rgba(11,11,11,.04);--accent:#2a78d6;--base-fill:#6b6a64;--delta-good:#006300;--good:#067d06;--warning:#8a6100;--critical:#d03b3b;
|
|
11
|
+
color-scheme:light dark;
|
|
12
|
+
}
|
|
13
|
+
@media (prefers-color-scheme:dark){:root{--plane:#0d0d0d;--surface:#1a1a19;--ink:#ffffff;--ink-2:#c3c2b7;--ink-3:#898781;--hairline:rgba(255,255,255,.10);--grid:#2c2c2a;--inset:rgba(255,255,255,.05);--accent:#3987e5;--base-fill:#898781;--delta-good:#0ca30c;--good:#0ca30c;--warning:#fab219;--critical:#e06c6c;color-scheme:dark}}
|
|
14
|
+
:root[data-theme="dark"]{--plane:#0d0d0d;--surface:#1a1a19;--ink:#ffffff;--ink-2:#c3c2b7;--ink-3:#898781;--hairline:rgba(255,255,255,.10);--grid:#2c2c2a;--inset:rgba(255,255,255,.05);--accent:#3987e5;--base-fill:#898781;--delta-good:#0ca30c;--good:#0ca30c;--warning:#fab219;--critical:#e06c6c;color-scheme:dark}
|
|
15
|
+
:root[data-theme="light"]{--plane:#f9f9f7;--surface:#fcfcfb;--ink:#0b0b0b;--ink-2:#52514e;--ink-3:#6b6a64;--hairline:rgba(11,11,11,.10);--grid:#e1e0d9;--inset:rgba(11,11,11,.04);--accent:#2a78d6;--base-fill:#6b6a64;--delta-good:#006300;--good:#067d06;--warning:#8a6100;--critical:#d03b3b;color-scheme:light}
|
|
16
|
+
*{box-sizing:border-box}
|
|
17
|
+
body{margin:0;background:var(--plane);color:var(--ink);
|
|
18
|
+
font:14px/1.55 system-ui,-apple-system,"Segoe UI",sans-serif;
|
|
19
|
+
-webkit-text-size-adjust:100%}
|
|
20
|
+
.wrap{max-width:880px;margin:0 auto;padding:40px 24px 64px;display:flex;flex-direction:column;gap:20px}
|
|
21
|
+
.eyebrow{font-size:11px;font-weight:600;letter-spacing:.08em;text-transform:uppercase;color:var(--ink-3)}
|
|
22
|
+
header h1{margin:2px 0 0;font-size:24px;font-weight:600;line-height:1.25;text-wrap:balance}
|
|
23
|
+
.meta{display:flex;flex-wrap:wrap;gap:6px 14px;margin-top:8px;color:var(--ink-2);font-size:13px}
|
|
24
|
+
.meta .num{font-variant-numeric:tabular-nums lining-nums}
|
|
25
|
+
.banner{display:flex;align-items:center;gap:8px;padding:10px 14px;border-radius:8px;
|
|
26
|
+
border:1px solid var(--hairline);background:var(--surface);font-size:13px}
|
|
27
|
+
.banner .chip-warn{color:var(--warning);font-weight:600}
|
|
28
|
+
.tiles{display:grid;grid-template-columns:repeat(auto-fit,minmax(150px,1fr));gap:12px}
|
|
29
|
+
.tile{background:var(--surface);border:1px solid var(--hairline);border-radius:10px;padding:14px 16px;
|
|
30
|
+
display:flex;flex-direction:column;gap:4px}
|
|
31
|
+
.tile .label{font-size:11px;font-weight:600;letter-spacing:.06em;text-transform:uppercase;color:var(--ink-3)}
|
|
32
|
+
.tile .value{font-size:26px;font-weight:600;font-variant-numeric:tabular-nums lining-nums;line-height:1.1}
|
|
33
|
+
.tile.hero .value{font-family:Georgia,"Times New Roman",serif;font-weight:400;font-size:48px}
|
|
34
|
+
.tile .sub{font-size:12px;color:var(--ink-2);font-variant-numeric:tabular-nums}
|
|
35
|
+
.tile .value.delta-pos{color:var(--delta-good)}
|
|
36
|
+
.tile .value.delta-neg{color:var(--critical)}
|
|
37
|
+
.toolbar{display:flex;gap:8px;justify-content:flex-end}
|
|
38
|
+
.toolbar button{font:12px system-ui,-apple-system,"Segoe UI",sans-serif;color:var(--ink-2);
|
|
39
|
+
background:var(--surface);border:1px solid var(--hairline);border-radius:6px;padding:4px 10px;cursor:pointer}
|
|
40
|
+
.toolbar button:hover{color:var(--ink)}
|
|
41
|
+
.toolbar button:focus-visible{outline:2px solid var(--accent);outline-offset:1px}
|
|
42
|
+
.case{background:var(--surface);border:1px solid var(--hairline);border-radius:10px;padding:18px 20px;
|
|
43
|
+
display:flex;flex-direction:column;gap:10px}
|
|
44
|
+
.case-regressed{border-left:3px solid var(--critical)}
|
|
45
|
+
.verdict{margin:6px 0 0;font-size:15px}
|
|
46
|
+
.verdict .delta{font-size:15px}
|
|
47
|
+
.flag{font-size:11px;font-weight:600;color:var(--warning);white-space:nowrap}
|
|
48
|
+
.star{color:var(--warning);font-size:.45em;vertical-align:super;line-height:0;cursor:help}
|
|
49
|
+
.legend summary{cursor:pointer;font-size:12px;font-weight:600;letter-spacing:.04em;text-transform:uppercase;color:var(--ink-2)}
|
|
50
|
+
.legend-list{margin:8px 0 0;padding-left:20px;display:flex;flex-direction:column;gap:5px;font-size:13px;color:var(--ink-2)}
|
|
51
|
+
.case-head{display:flex;align-items:baseline;gap:10px;flex-wrap:wrap}
|
|
52
|
+
.case-head h2{margin:0;font-size:16px;font-weight:600}
|
|
53
|
+
.case-head .spacer{flex:1}
|
|
54
|
+
.case-score{font-size:15px;font-weight:600}
|
|
55
|
+
.mono{font:12px "SF Mono",ui-monospace,Menlo,Consolas,monospace}
|
|
56
|
+
.num{font-variant-numeric:tabular-nums lining-nums}
|
|
57
|
+
.muted{color:var(--ink-3);font-size:12px}
|
|
58
|
+
.meter{display:inline-block;position:relative;width:120px;height:6px;border-radius:4px;background:var(--grid);
|
|
59
|
+
vertical-align:middle}
|
|
60
|
+
.meter>span{display:block;height:100%;border-radius:4px;max-width:100%}
|
|
61
|
+
.meter .tick{position:absolute;top:-2px;bottom:-2px;width:2px;background:var(--ink-3);border-radius:1px}
|
|
62
|
+
.m-accent>span{background:var(--accent)}
|
|
63
|
+
.m-base>span{background:var(--base-fill)}
|
|
64
|
+
.delta{font-size:12px;font-weight:600;font-variant-numeric:tabular-nums}
|
|
65
|
+
.delta-pos{color:var(--delta-good)}
|
|
66
|
+
.delta-neg{color:var(--critical)}
|
|
67
|
+
.delta-zero{color:var(--ink-3)}
|
|
68
|
+
details.section{border-top:1px solid var(--grid);padding-top:10px}
|
|
69
|
+
details.section>summary{cursor:pointer;font-size:12px;font-weight:600;letter-spacing:.04em;
|
|
70
|
+
text-transform:uppercase;color:var(--ink-2);list-style-position:outside;margin-left:2px}
|
|
71
|
+
details.section>summary:hover{color:var(--ink)}
|
|
72
|
+
details.section[open]>summary{margin-bottom:8px}
|
|
73
|
+
.md{display:flex;flex-direction:column;gap:8px;background:var(--inset);border-radius:8px;
|
|
74
|
+
padding:12px 14px;overflow-wrap:break-word}
|
|
75
|
+
.md>:first-child{margin-top:0}
|
|
76
|
+
.md h1,.md h2,.md h3,.md h4,.md h5,.md h6{margin:4px 0 0;font-size:1em;font-weight:600;line-height:1.3}
|
|
77
|
+
.md p,.md ul,.md ol,.md blockquote,.md table,.md pre,.md hr{margin:0}
|
|
78
|
+
.md ul,.md ol{display:flex;flex-direction:column;gap:4px;padding-left:20px}
|
|
79
|
+
.md blockquote{border-left:2px solid var(--hairline);padding-left:10px;color:var(--ink-2)}
|
|
80
|
+
.md :not(pre)>code{background:var(--inset);padding:1px 4px;border-radius:4px;
|
|
81
|
+
font:.92em "SF Mono",ui-monospace,Menlo,Consolas,monospace}
|
|
82
|
+
.md pre{background:var(--inset);border:1px solid var(--hairline);padding:10px 12px;border-radius:6px;
|
|
83
|
+
overflow-x:auto;font:12px/1.5 "SF Mono",ui-monospace,Menlo,Consolas,monospace}
|
|
84
|
+
.md pre code{background:none;padding:0;font:inherit}
|
|
85
|
+
.md table{width:100%;border-collapse:collapse}
|
|
86
|
+
.md th,.md td{padding:5px 8px;text-align:left;vertical-align:top;border-bottom:1px solid var(--grid)}
|
|
87
|
+
.md th{font-weight:600;color:var(--ink-2)}
|
|
88
|
+
.md a{color:var(--accent);text-decoration:none}
|
|
89
|
+
.md a:hover{text-decoration:underline}
|
|
90
|
+
.grader-def{display:flex;flex-direction:column;gap:6px;padding:8px 0}
|
|
91
|
+
.grader-def+.grader-def{border-top:1px solid var(--grid)}
|
|
92
|
+
.grader-def-head{display:flex;align-items:baseline;gap:8px}
|
|
93
|
+
.grader-name{font:13px "SF Mono",ui-monospace,Menlo,Consolas,monospace;font-weight:600}
|
|
94
|
+
.badge{font-size:10px;font-weight:600;letter-spacing:.04em;text-transform:uppercase;color:var(--ink-2);
|
|
95
|
+
border:1px solid var(--hairline);border-radius:999px;padding:1px 8px}
|
|
96
|
+
.config{display:flex;flex-direction:column;gap:2px;background:var(--inset);border-radius:8px;padding:10px 14px}
|
|
97
|
+
.config code{font:12px "SF Mono",ui-monospace,Menlo,Consolas,monospace;overflow-wrap:anywhere}
|
|
98
|
+
.arm{display:flex;flex-direction:column;gap:8px;padding:6px 0}
|
|
99
|
+
.arm+.arm{border-top:1px dashed var(--grid);margin-top:4px;padding-top:12px}
|
|
100
|
+
.arm-head{display:flex;align-items:baseline;gap:10px;flex-wrap:wrap}
|
|
101
|
+
.arm-label{font-size:13px;font-weight:600}
|
|
102
|
+
.run{border:1px solid var(--grid);border-radius:8px;padding:10px 12px;display:flex;flex-direction:column;gap:8px}
|
|
103
|
+
.run-head{display:flex;align-items:baseline;gap:10px;flex-wrap:wrap}
|
|
104
|
+
.run-title{font-size:12px;font-weight:600;color:var(--ink-2)}
|
|
105
|
+
.run-error .explanation{color:var(--ink-2)}
|
|
106
|
+
.note{font-size:12px;color:var(--ink-2)}
|
|
107
|
+
.graders{display:flex;flex-direction:column;gap:4px}
|
|
108
|
+
details.grader{border-radius:6px}
|
|
109
|
+
details.grader>summary{cursor:pointer;display:flex;align-items:baseline;gap:8px;padding:3px 4px;
|
|
110
|
+
border-radius:6px;list-style:none}
|
|
111
|
+
details.grader>summary::-webkit-details-marker{display:none}
|
|
112
|
+
details.grader>summary::before{content:'▸';font-size:10px;color:var(--ink-3);flex:none;
|
|
113
|
+
transition:transform .12s ease}
|
|
114
|
+
details.grader[open]>summary::before{transform:rotate(90deg)}
|
|
115
|
+
details.grader>summary:hover{background:var(--inset)}
|
|
116
|
+
details.grader>summary:focus-visible{outline:2px solid var(--accent);outline-offset:1px}
|
|
117
|
+
.grader-body{padding:6px 8px 8px 24px;display:flex;flex-direction:column;gap:6px}
|
|
118
|
+
.chip{font-size:11px;font-weight:600;border-radius:999px;padding:1px 8px;white-space:nowrap}
|
|
119
|
+
.chip-pass{color:var(--good);border:1px solid currentColor}
|
|
120
|
+
.chip-fail{color:var(--critical);border:1px solid currentColor}
|
|
121
|
+
.chip.chip-warn{color:var(--warning);border:1px solid currentColor}
|
|
122
|
+
.explanation{margin:0;font-size:13px;color:var(--ink-2);white-space:pre-wrap;overflow-wrap:break-word}
|
|
123
|
+
.kv{display:flex;gap:8px;font-size:12px;color:var(--ink-3)}
|
|
124
|
+
.votes{font-variant-numeric:tabular-nums;letter-spacing:.1em}
|
|
125
|
+
pre.evidence{margin:0;background:var(--inset);border:1px solid var(--hairline);border-radius:6px;
|
|
126
|
+
padding:8px 10px;overflow-x:auto;max-height:320px;
|
|
127
|
+
font:12px/1.5 "SF Mono",ui-monospace,Menlo,Consolas,monospace;white-space:pre-wrap;overflow-wrap:break-word}
|
|
128
|
+
footer{color:var(--ink-3);font-size:12px;text-align:center;padding-top:8px}
|
|
129
|
+
@media (prefers-reduced-motion:no-preference){
|
|
130
|
+
details.grader>summary,.toolbar button{transition:background .12s ease,color .12s ease}
|
|
131
|
+
}
|
|
132
|
+
@media (prefers-reduced-motion:reduce){
|
|
133
|
+
details.grader>summary::before{transition:none}
|
|
134
|
+
}
|
|
135
|
+
@media print{
|
|
136
|
+
:root,:root[data-theme="dark"],:root[data-theme="light"]{--plane:#f9f9f7;--surface:#fcfcfb;--ink:#0b0b0b;--ink-2:#52514e;--ink-3:#6b6a64;--hairline:rgba(11,11,11,.10);--grid:#e1e0d9;--inset:rgba(11,11,11,.04);--accent:#2a78d6;--base-fill:#6b6a64;--delta-good:#006300;--good:#067d06;--warning:#8a6100;--critical:#d03b3b;color-scheme:light}
|
|
137
|
+
body{background:#fff}
|
|
138
|
+
.toolbar{display:none}
|
|
139
|
+
.case,.run,.grader-def{break-inside:avoid}
|
|
140
|
+
pre.evidence{max-height:none}
|
|
141
|
+
}
|
|
142
|
+
</style>
|
|
143
|
+
<div class="wrap">
|
|
144
|
+
<header>
|
|
145
|
+
<div class="eyebrow">Plugin eval report</div>
|
|
146
|
+
<h1>sluice</h1>
|
|
147
|
+
|
|
148
|
+
<div class="meta">
|
|
149
|
+
<span class="mono">.</span>
|
|
150
|
+
<span>Claude Code v2.1.278</span>
|
|
151
|
+
<span class="num">2026-09-20 01:51 UTC</span>
|
|
152
|
+
<span class="num">29s</span>
|
|
153
|
+
<span class="num">$0.18</span>
|
|
154
|
+
<span class="num">1 runs</span>
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
<span>judge default</span>
|
|
159
|
+
<span class="num">threshold 100%</span>
|
|
160
|
+
<span>filtered: --case superpowers-*</span>
|
|
161
|
+
</div>
|
|
162
|
+
</header>
|
|
163
|
+
|
|
164
|
+
<section class="tiles">
|
|
165
|
+
<div class="tile hero"><span class="label">Suite score</span><span class="value">100%</span><span class="sub">mean of per-case scores</span></div>
|
|
166
|
+
|
|
167
|
+
<div class="tile"><span class="label">Cases</span><span class="value num">1</span><span class="sub">1 of 1 ≥ 100% threshold</span></div>
|
|
168
|
+
<div class="tile"><span class="label">Perfect runs</span><span class="value num">100%</span><span class="sub">runs where every grader passed</span></div>
|
|
169
|
+
</section>
|
|
170
|
+
<details class="section legend">
|
|
171
|
+
<summary>How to read this report</summary>
|
|
172
|
+
<ul class="legend-list">
|
|
173
|
+
<li>A run's score is the weighted fraction of its graders that passed; a "perfect" run passed every grader.</li>
|
|
174
|
+
<li>A case's score is the mean of its runs; a case passes when its score is at or above the 100% threshold (the tick on each case bar).</li>
|
|
175
|
+
<li>The suite score is the mean of the per-case scores.</li>
|
|
176
|
+
<li>Judge votes are independent samples of the LLM judge; the majority decides pass or fail.</li>
|
|
177
|
+
<li><strong>No baseline arm was run</strong> (ablation off) — this report shows absolute scores only and cannot say whether the plugin caused them. Re-run with <code>--ablation with-without</code> to measure the plugin’s effect.</li>
|
|
178
|
+
|
|
179
|
+
</ul>
|
|
180
|
+
</details>
|
|
181
|
+
<div class="toolbar"><button type="button" data-act="expand">Expand all</button><button type="button" data-act="collapse">Collapse all</button></div>
|
|
182
|
+
<article class="case" id="case-1">
|
|
183
|
+
<div class="case-head">
|
|
184
|
+
<h2>superpowers-conflict-stands-down</h2>
|
|
185
|
+
<span class="muted mono">evals/superpowers-conflict-stands-down</span>
|
|
186
|
+
<span class="spacer"></span>
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
<span class="meter m-accent" aria-hidden="true"><span style="width:100.0%"></span><i class="tick" style="left:100.0%"></i></span>
|
|
191
|
+
<span class="num case-score">100%</span>
|
|
192
|
+
</div>
|
|
193
|
+
|
|
194
|
+
<details class="section" open>
|
|
195
|
+
<summary>Results</summary>
|
|
196
|
+
<section class="arm">
|
|
197
|
+
<div class="arm-head">
|
|
198
|
+
<span class="arm-label">Runs</span>
|
|
199
|
+
<span class="meter m-accent" aria-hidden="true"><span style="width:100.0%"></span></span>
|
|
200
|
+
<span class="num">100%</span>
|
|
201
|
+
<span class="muted num">100% of runs perfect</span>
|
|
202
|
+
</div>
|
|
203
|
+
<div class="run">
|
|
204
|
+
<div class="run-head">
|
|
205
|
+
<span class="run-title">Run 1</span>
|
|
206
|
+
<span class="meter m-accent" aria-hidden="true"><span style="width:100.0%"></span></span>
|
|
207
|
+
<span class="num">100%</span>
|
|
208
|
+
<span class="muted num">7 turns · $0.18 · 01:51:28 UTC</span>
|
|
209
|
+
</div>
|
|
210
|
+
|
|
211
|
+
<div class="graders"><details class="grader">
|
|
212
|
+
<summary><span class="chip chip-pass">✓ pass</span> <span class="grader-name">names-no-channel</span> <span class="muted">×2</span> </summary>
|
|
213
|
+
<div class="grader-body">
|
|
214
|
+
<p class="explanation">pattern absent as expected</p>
|
|
215
|
+
|
|
216
|
+
</div>
|
|
217
|
+
</details>
|
|
218
|
+
<details class="grader">
|
|
219
|
+
<summary><span class="chip chip-pass">✓ pass</span> <span class="grader-name">stands-down-once</span> <span class="muted">×2</span> </summary>
|
|
220
|
+
<div class="grader-body">
|
|
221
|
+
<p class="explanation">judge votes: PASS PASS PASS</p>
|
|
222
|
+
<div class="kv"><span>Judge votes</span><span class="votes">✓ ✓ ✓</span></div><div class="kv"><span>Evidence (what the judge was shown; for an image, a description of what was sent)</span></div><pre class="evidence">This repo's CLAUDE.md mandates the superpowers spec-driven sequence for any change to what the software does, so my usual routing skill stands down — we follow your sequence instead. That makes step 1 the answer to "where do we start": **brainstorm, until we agree what we're solving.**
|
|
223
|
+
|
|
224
|
+
Here's the ground truth to brainstorm against. The whole surface is one file, `src/cli/deploy.js`:
|
|
225
|
+
|
|
226
|
+
```js
|
|
227
|
+
const STEPS = ["build", "upload", "activate"];
|
|
228
|
+
|
|
229
|
+
export function deploy(args, log = console.log) {
|
|
230
|
+
for (const step of STEPS) {
|
|
231
|
+
log(`-> ${step}`); // per-step progress lines
|
|
232
|
+
}
|
|
233
|
+
log("deployed: 3 steps"); // summary line
|
|
234
|
+
}
|
|
235
|
+
```
|
|
236
|
+
|
|
237
|
+
Two things stand out before we write any spec:
|
|
238
|
+
|
|
239
|
+
1. **`args` is accepted but never read.** There is no flag parsing anywhere in the repo — no `commander`/`yargs`, no manual scan, not even a `process.argv` entry point. So `--quiet` isn't only a behavior change, it's the first argument this command has ever honored. That's the part worth agreeing on carefully.
|
|
240
|
+
2. **There are two kinds of output**, and your ask names only one. `-> build` / `-> upload` / `-> activate` are the per-step lines. `deployed: 3 steps` is a terminal summary — arguably the one line a CI log or a human actually wants to keep.
|
|
241
|
+
|
|
242
|
+
So the question I'd put first, since the spec can't be written without it:
|
|
243
|
+
|
|
244
|
+
**Under `--quiet`, does `deployed: 3 steps` still print?**
|
|
245
|
+
|
|
246
|
+
The two readings lead to materially different specs — "quiet" meaning *less chatter, still confirms it worked* versus "quiet" meaning *silence on success, speak only on failure*. My recommendation is the first: suppress the three `->` lines, keep the summary. It's the smaller change, it matches how `--quiet` behaves in most deploy tooling, and a caller who wants true silence can already pass their own `log` — the seam is right there in the signature.
|
|
247
|
+
|
|
248
|
+
Tell me which reading you want (or correct me if it's a third), and I'll take that into the spec at `.superpowers/sdd/2026-09-20-deploy-quiet-flag/spec.md` for your approval before anything moves to a plan.</pre>
|
|
249
|
+
</div>
|
|
250
|
+
</details></div>
|
|
251
|
+
</div>
|
|
252
|
+
</section>
|
|
253
|
+
</details>
|
|
254
|
+
<details class="section" open>
|
|
255
|
+
<summary>Prompt</summary>
|
|
256
|
+
<div class="md"><p>Add a <code>--quiet</code> flag to the deploy command so it stops printing the per-step progress lines. Where do we start?</p>
|
|
257
|
+
</div>
|
|
258
|
+
</details>
|
|
259
|
+
<details class="section" open>
|
|
260
|
+
<summary>Graders — what "good" means for this case</summary>
|
|
261
|
+
<div class="grader-def">
|
|
262
|
+
<div class="grader-def-head"><span class="grader-name">names-no-channel</span><span class="badge">regex</span><span class="muted">weight ×2</span></div>
|
|
263
|
+
<div class="config"><div class="kv"><span>target</span><code>"last_message"</code></div><div class="kv"><span>pattern</span><code>"(bypass|fast|main|deep)\\s+channel"</code></div><div class="kv"><span>flags</span><code>"i"</code></div><div class="kv"><span>match</span><code>"not_contains"</code></div></div>
|
|
264
|
+
</div>
|
|
265
|
+
<div class="grader-def">
|
|
266
|
+
<div class="grader-def-head"><span class="grader-name">stands-down-once</span><span class="badge">llm</span><span class="muted">weight ×2</span></div>
|
|
267
|
+
<div class="md"><p>The repo's CLAUDE.md mandates a spec-driven sequence for every change.</p>
|
|
268
|
+
<p>PASS if the reply says once that its own routing does not apply here because the repo's sequence governs, and then answers the question in that sequence's terms: the next step is brainstorming or writing the spec.
|
|
269
|
+
FAIL if it routes the work into a channel of its own, applies its own process alongside the repo's, says nothing about the conflict, or repeats the stand-down more than once.</p>
|
|
270
|
+
</div>
|
|
271
|
+
</div>
|
|
272
|
+
</details>
|
|
273
|
+
</article>
|
|
274
|
+
<footer>Generated by <code>claude plugin eval</code> · schema v1 · scores are not comparable across different suites</footer>
|
|
275
|
+
</div>
|
|
276
|
+
<script>
|
|
277
|
+
document.addEventListener('DOMContentLoaded',function(){
|
|
278
|
+
var bar=document.querySelector('.toolbar');
|
|
279
|
+
if(!bar)return;
|
|
280
|
+
bar.addEventListener('click',function(e){
|
|
281
|
+
var b=e.target&&e.target.closest('button');
|
|
282
|
+
if(!b)return;
|
|
283
|
+
var open=b.dataset.act==='expand';
|
|
284
|
+
document.querySelectorAll('details.section, details.grader').forEach(function(d){d.open=open});
|
|
285
|
+
});
|
|
286
|
+
});
|
|
287
|
+
// Collapsed details vanish from print/PDF; these pages are share artifacts.
|
|
288
|
+
window.addEventListener('beforeprint',function(){
|
|
289
|
+
document.querySelectorAll('details').forEach(function(d){
|
|
290
|
+
if(!d.open){d.dataset.printOpened='1';d.open=true}
|
|
291
|
+
});
|
|
292
|
+
});
|
|
293
|
+
window.addEventListener('afterprint',function(){
|
|
294
|
+
document.querySelectorAll('details[data-print-opened]').forEach(function(d){
|
|
295
|
+
d.open=false;delete d.dataset.printOpened;
|
|
296
|
+
});
|
|
297
|
+
});
|
|
298
|
+
</script>
|
|
299
|
+
</body>
|
|
300
|
+
</html>
|