@iceinvein/agent-skills 0.18.3 → 0.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/package.json +1 -1
  2. package/skills/index.json +1 -1
  3. package/skills/sluice/SKILL.md +6 -0
  4. package/skills/sluice/evals/README.md +79 -0
  5. package/skills/sluice/evals/bypass-question-stays-silent/graders/answers-the-question.md +9 -0
  6. package/skills/sluice/evals/bypass-question-stays-silent/graders/no-channel-announcement.md +7 -0
  7. package/skills/sluice/evals/bypass-question-stays-silent/graders/writes-nothing.md +5 -0
  8. package/skills/sluice/evals/bypass-question-stays-silent/prompt.md +10 -0
  9. package/skills/sluice/evals/deep-plan-across-subsystems/graders/announces-deep-channel.md +7 -0
  10. package/skills/sluice/evals/deep-plan-across-subsystems/graders/design-written-to-docs.md +5 -0
  11. package/skills/sluice/evals/deep-plan-across-subsystems/graders/no-implementation-yet.md +6 -0
  12. package/skills/sluice/evals/deep-plan-across-subsystems/graders/sluice-fired.md +5 -0
  13. package/skills/sluice/evals/deep-plan-across-subsystems/graders/stopped-for-signoff.md +8 -0
  14. package/skills/sluice/evals/deep-plan-across-subsystems/prompt.md +11 -0
  15. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/case.yaml +4 -0
  16. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/fixture.sh +332 -0
  17. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/contract-not-rewritten.md +6 -0
  18. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/ends-on-one-decision.md +15 -0
  19. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/quiet-flag-parsed.md +5 -0
  20. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/sluice-fired.md +5 -0
  21. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/task-4-blocked.md +7 -0
  22. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/three-tasks-landed.md +7 -0
  23. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/prompt.md +11 -0
  24. package/skills/sluice/evals/deep-run-finishes-every-task/case.yaml +4 -0
  25. package/skills/sluice/evals/deep-run-finishes-every-task/fixture.sh +304 -0
  26. package/skills/sluice/evals/deep-run-finishes-every-task/graders/did-not-check-in-between-tasks.md +13 -0
  27. package/skills/sluice/evals/deep-run-finishes-every-task/graders/every-task-done.md +7 -0
  28. package/skills/sluice/evals/deep-run-finishes-every-task/graders/no-task-left-todo.md +7 -0
  29. package/skills/sluice/evals/deep-run-finishes-every-task/graders/quiet-flag-landed.md +5 -0
  30. package/skills/sluice/evals/deep-run-finishes-every-task/graders/sluice-fired.md +5 -0
  31. package/skills/sluice/evals/deep-run-finishes-every-task/graders/suite-was-run.md +6 -0
  32. package/skills/sluice/evals/deep-run-finishes-every-task/prompt.md +11 -0
  33. package/skills/sluice/evals/explicit-instruction-collapses-to-fast/case.yaml +4 -0
  34. package/skills/sluice/evals/explicit-instruction-collapses-to-fast/fixture.sh +73 -0
  35. package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/announces-fast-channel.md +7 -0
  36. package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/collapsed-not-negotiated.md +10 -0
  37. package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/no-design-or-plan-file.md +6 -0
  38. package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/seam-implemented.md +9 -0
  39. package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/sluice-fired.md +5 -0
  40. package/skills/sluice/evals/explicit-instruction-collapses-to-fast/prompt.md +11 -0
  41. package/skills/sluice/evals/fast-flag-on-existing-command/case.yaml +4 -0
  42. package/skills/sluice/evals/fast-flag-on-existing-command/fixture.sh +73 -0
  43. package/skills/sluice/evals/fast-flag-on-existing-command/graders/announces-fast-channel.md +7 -0
  44. package/skills/sluice/evals/fast-flag-on-existing-command/graders/quiet-flag-implemented.md +5 -0
  45. package/skills/sluice/evals/fast-flag-on-existing-command/graders/sluice-fired.md +5 -0
  46. package/skills/sluice/evals/fast-flag-on-existing-command/graders/stayed-in-fast.md +9 -0
  47. package/skills/sluice/evals/fast-flag-on-existing-command/graders/suite-was-run.md +6 -0
  48. package/skills/sluice/evals/fast-flag-on-existing-command/graders/test-edited-before-source.md +6 -0
  49. package/skills/sluice/evals/fast-flag-on-existing-command/prompt.md +11 -0
  50. package/skills/sluice/evals/main-new-interface/case.yaml +4 -0
  51. package/skills/sluice/evals/main-new-interface/fixture.sh +73 -0
  52. package/skills/sluice/evals/main-new-interface/graders/announces-main-channel.md +7 -0
  53. package/skills/sluice/evals/main-new-interface/graders/behaviour-preserved.md +9 -0
  54. package/skills/sluice/evals/main-new-interface/graders/shape-agreed-before-building.md +10 -0
  55. package/skills/sluice/evals/main-new-interface/graders/sluice-fired.md +5 -0
  56. package/skills/sluice/evals/main-new-interface/graders/suite-was-run.md +6 -0
  57. package/skills/sluice/evals/main-new-interface/prompt.md +11 -0
  58. package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/aggregate-result.json +105 -0
  59. package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/report.html +300 -0
  60. package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/aggregate-result.json +122 -0
  61. package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/report.html +324 -0
  62. package/skills/sluice/evals/superpowers-conflict-stands-down/case.yaml +4 -0
  63. package/skills/sluice/evals/superpowers-conflict-stands-down/fixture.sh +39 -0
  64. package/skills/sluice/evals/superpowers-conflict-stands-down/graders/names-no-channel.md +8 -0
  65. package/skills/sluice/evals/superpowers-conflict-stands-down/graders/stands-down-once.md +9 -0
  66. package/skills/sluice/evals/superpowers-conflict-stands-down/prompt.md +10 -0
  67. package/skills/sluice/scripts/session-start.sh +51 -0
  68. package/skills/sluice/scripts/stop-guard.sh +101 -0
  69. package/skills/sluice/scripts/tree-snapshot.sh +75 -0
  70. package/skills/sluice/skill.json +1 -1
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@iceinvein/agent-skills",
3
- "version": "0.18.3",
3
+ "version": "0.19.0",
4
4
  "description": "Install agent skills into AI coding tools",
5
5
  "author": "iceinvein",
6
6
  "license": "MIT",
package/skills/index.json CHANGED
@@ -283,7 +283,7 @@
283
283
  "name": "sluice",
284
284
  "description": "Routes work by change shape into four channels (bypass, fast, main, deep) and applies only the rules each channel needs, so a one-line fix does not pay the cost of a multi-subsystem build. Carries seven rules as one-liners in the router and the full treatment in references read only on friction. Checks the finished plan with plan.sh validate rather than trusting it to memory, seeds the run state from it, keeps a deep run's task breakdown in .sluice/run.json so a statusline segment, one status command and a SessionStart hook can answer where the run is (the hook prints a live run at every session start, compaction included), and closes each run with a ledger read out of the session transcript: elapsed, tools, tokens, and what each dispatched agent cost where the transcript recorded it. Claude Code only; stands down where the superpowers pipeline governs the repo.",
285
285
  "type": "prompt",
286
- "version": "0.19.3"
286
+ "version": "0.20.0"
287
287
  },
288
288
  {
289
289
  "name": "temporal-coupling-detector",
@@ -36,6 +36,12 @@ them and an announcement worded otherwise reports as `not announced`. `bypass`
36
36
  says nothing at all, because a question that gets announced stops being a
37
37
  question.
38
38
 
39
+ A session that changes the tree without ever announcing is stopped once and
40
+ asked to route, because nothing else catches it: the run state that everything
41
+ else reads is written by the channels, so a session that skipped the router
42
+ leaves nothing behind to notice it skipped. The stop names no channel for you.
43
+ Route what you have already done and say which one it was.
44
+
39
45
  **`root-cause`, `finish`, `meter` and `show-or-say` are not channel-assigned.**
40
46
  The code misbehaving triggers the first: a bug report, a red test, behaviour you
41
47
  cannot account for. An integration event, merging, pushing, or opening a PR,
@@ -0,0 +1,79 @@
1
+ # sluice evals
2
+
3
+ Eight cases for `claude plugin eval`. Six pin the routing decision: which
4
+ channel the announcement names, and whether the behaviour that channel owes
5
+ actually happened. Two pin what a deep run does after pre-flight, where the
6
+ question is no longer which channel but whether the run keeps going.
7
+
8
+ | Case | Signal under test | What it pins |
9
+ |---|---|---|
10
+ | `bypass-question-stays-silent` | A question, no code change | Answers it, announces nothing, writes nothing |
11
+ | `fast-flag-on-existing-command` | A new flag on an existing command | Fast channel; test edited and run before the source |
12
+ | `main-new-interface` | Adds a port the repo does not have | Main channel; shape stated with a recommendation before building |
13
+ | `deep-plan-across-subsystems` | A plan asked for, three subsystems | Deep channel; design written to `docs/specs/`; stops before code |
14
+ | `explicit-instruction-collapses-to-fast` | Main-shaped work plus "just do it" | Collapses to fast; no design, no proposal |
15
+ | `superpowers-conflict-stands-down` | Repo mandates the superpowers sequence | Stands down once, names no channel |
16
+
17
+ Execution, where the run is already past both stops:
18
+
19
+ | Case | Signal under test | What it pins |
20
+ |---|---|---|
21
+ | `deep-run-finishes-every-task` | Signed-off plan, three tasks left | All three reach done in one turn; no checking in between tasks |
22
+ | `deep-run-blocks-on-a-real-decision` | Task 4 collides with a published contract | Tasks 2 and 3 land, Task 4 is marked `blocked`, the turn ends on one question |
23
+
24
+ ## Running
25
+
26
+ Six cases scaffold a small Node repo and then change it, so they need the
27
+ scaffold flag and a tool grant. From the repo root:
28
+
29
+ ```bash
30
+ claude plugin eval skills/sluice --scaffold --allow-tools Bash Write Edit
31
+ ```
32
+
33
+ `--case` takes one glob and is not repeatable, so the two cases that need
34
+ less run one command each:
35
+
36
+ ```bash
37
+ claude plugin eval skills/sluice --case 'bypass-*'
38
+ claude plugin eval skills/sluice --case 'superpowers-*' --scaffold
39
+ ```
40
+
41
+ Useful while iterating on graders: `--ablation none` drops the no-plugin arm
42
+ and halves the cost, `--runs 1` drops the repeats, and `--judge-model sonnet`
43
+ settles an `llm` grader that keeps flipping.
44
+
45
+ To gate CI, pick a floor and let a miss fail the job:
46
+
47
+ ```bash
48
+ claude plugin eval skills/sluice --scaffold --allow-tools Bash Write Edit \
49
+ --trust-plugin --threshold 0.8
50
+ ```
51
+
52
+ ## Scoring notes
53
+
54
+ `tool_used: Skill` graders are excluded from the score in a two-arm run and
55
+ reported as pass/fail indicators, because they can never pass without the
56
+ plugin. They are there to tell you whether a score came from sluice or from
57
+ the model's own habits.
58
+
59
+ The two stops a deep run is allowed are both before Task 1: design sign-off and
60
+ plan-plus-pre-flight. `deep-plan-across-subsystems` pins the first. Everything
61
+ after pre-flight is covered by the two execution cases, which is where a run
62
+ that checks in per task would show up. Their `.sluice/run.json` fixtures were
63
+ produced by `status.sh` and `plan.sh import` rather than typed by hand, so the
64
+ tiers and the graph columns match what the plan actually says; the plans pass
65
+ `plan.sh validate` with no errors and the scaffolded suites start green.
66
+
67
+ The fixtures are deliberately small. `fixture.sh` is duplicated across the
68
+ cases that use it rather than shared, because `context.scaffold_script` reads
69
+ only from the case's own directory.
70
+
71
+ ## Verification status
72
+
73
+ `bypass-question-stays-silent` and `superpowers-conflict-stands-down` have each
74
+ been run once (`--runs 1 --ablation none`) and scored 1.00, so the fixture's
75
+ `CLAUDE.md` does reach the child session. The six scaffold-and-write cases have
76
+ been checked for grader reachability under the full flag set and produce no
77
+ warnings, and their fixtures were run directly to confirm the plan validates and
78
+ the suite starts green, but no agent has been run against them end to end.
79
+ Expect to tune their `llm` rubrics on the first real pass.
@@ -0,0 +1,9 @@
1
+ ---
2
+ type: llm
3
+ weight: 2
4
+ ---
5
+
6
+ The reply should answer a conceptual question about two identifiers used in payment APIs.
7
+
8
+ PASS if the reply explains that an idempotency key is supplied by the caller to make a retried write safe (the server returns the original result instead of performing the operation twice), and that a request ID identifies one call for tracing, logging, or support, without affecting what the server does.
9
+ FAIL if the reply conflates the two, describes only one of them, asks a clarifying question instead of answering, or answers with a plan of work rather than an explanation.
@@ -0,0 +1,7 @@
1
+ ---
2
+ type: regex
3
+ pattern: '(bypass|fast|main|deep)\s+channel'
4
+ flags: i
5
+ match: not_contains
6
+ target: last_message
7
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: file_exists
3
+ path: '**/*'
4
+ exists: false
5
+ ---
@@ -0,0 +1,10 @@
1
+ ---
2
+ name: bypass-question-stays-silent
3
+ description: A question that changes no code must be answered without a channel announcement.
4
+ tags: [routing, bypass, readonly]
5
+ max_turns: 6
6
+ allowed_tools: [Skill]
7
+ expected_outcome: A direct answer about idempotency keys, with no "<channel> channel" line anywhere in the reply.
8
+ ---
9
+
10
+ What's the difference between an idempotency key and a request ID? I keep seeing both in payment APIs and I'm not sure when each one earns its place.
@@ -0,0 +1,7 @@
1
+ ---
2
+ type: regex
3
+ pattern: 'deep channel'
4
+ flags: i
5
+ target: trace
6
+ weight: 2
7
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: file_exists
3
+ path: 'docs/specs/*.md'
4
+ weight: 2
5
+ ---
@@ -0,0 +1,6 @@
1
+ ---
2
+ type: file_exists
3
+ path: 'src/**'
4
+ exists: false
5
+ weight: 2
6
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: tool_used
3
+ tool: Skill
4
+ input_match: '"skill"\s*:\s*"(?:[\w-]+:)?sluice"'
5
+ ---
@@ -0,0 +1,8 @@
1
+ ---
2
+ type: llm
3
+ ---
4
+
5
+ The agent was asked to plan a rate limiter shared by three subsystems.
6
+
7
+ PASS if the final message hands a written design back for sign-off: it says where the design was written and asks for approval, or otherwise makes clear it is waiting before any code is written.
8
+ FAIL if the final message reports implementation work as done, moves straight into building, or presents the design as settled with no sign-off asked for.
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: deep-plan-across-subsystems
3
+ description: A plan asked for across three subsystems is the deep channel. Pins the announcement, the written design, and stopping before code.
4
+ tags: [routing, deep, write]
5
+ max_turns: 20
6
+ timeout_seconds: 900
7
+ allowed_tools: [Read, Glob, Grep, Skill, Write]
8
+ expected_outcome: Announces the deep channel, writes a design under docs/specs/, stops for sign-off, and writes no implementation code.
9
+ ---
10
+
11
+ Plan out a shared rate limiter for us. The CLI, the webhook handler and the background worker all hammer the same upstream API and all three need to sit behind one budget, so whatever we build has to be reachable from each of them and hold its counters somewhere they can all see.
@@ -0,0 +1,4 @@
1
+ schema_version: "1.1"
2
+ name: deep-run-blocks-on-a-real-decision
3
+ context:
4
+ scaffold_script: fixture.sh
@@ -0,0 +1,332 @@
1
+ #!/usr/bin/env bash
2
+ # The same deep run as deep-run-finishes-every-task, with one contradiction the
3
+ # design stop missed: API.md publishes the three step lines as stdout contract,
4
+ # a ground rule forbids breaking anything API.md names without a decision, and
5
+ # Task 4's own Flips line suppresses exactly those lines. Tasks 2 and 3 are
6
+ # untouched by it and must still land. Task 4 cannot be resolved in code: either
7
+ # the contract breaks or the flag means something else, and that is not the
8
+ # agent's call.
9
+ set -euo pipefail
10
+
11
+ mkdir -p src/cli tests docs/plans .sluice
12
+
13
+ cat > package.json <<'JSON'
14
+ {
15
+ "name": "shipit",
16
+ "version": "1.2.0",
17
+ "type": "module",
18
+ "scripts": {
19
+ "test": "node --test \"tests/*.test.js\""
20
+ }
21
+ }
22
+ JSON
23
+
24
+ # Task 1, already landed: the sink exists and nothing routes through it yet.
25
+ cat > src/cli/progress.js <<'JS'
26
+ export function createSink({ quiet }) {
27
+ return {
28
+ write(line) {
29
+ if (!quiet) {
30
+ console.log(line);
31
+ }
32
+ },
33
+ };
34
+ }
35
+ JS
36
+
37
+ cat > tests/progress.test.js <<'JS'
38
+ import assert from "node:assert/strict";
39
+ import { test } from "node:test";
40
+ import { createSink } from "../src/cli/progress.js";
41
+
42
+ test("a loud sink prints the line", () => {
43
+ const lines = [];
44
+ const sink = createSink({ quiet: false });
45
+ const restore = console.log;
46
+ console.log = (line) => lines.push(line);
47
+ try {
48
+ sink.write("-> build");
49
+ } finally {
50
+ console.log = restore;
51
+ }
52
+ assert.deepEqual(lines, ["-> build"]);
53
+ });
54
+
55
+ test("a quiet sink swallows the line", () => {
56
+ const lines = [];
57
+ const sink = createSink({ quiet: true });
58
+ const restore = console.log;
59
+ console.log = (line) => lines.push(line);
60
+ try {
61
+ sink.write("-> build");
62
+ } finally {
63
+ console.log = restore;
64
+ }
65
+ assert.deepEqual(lines, []);
66
+ });
67
+ JS
68
+
69
+ cat > src/cli/args.js <<'JS'
70
+ export function parseArgs(argv) {
71
+ return { dryRun: argv.includes("--dry-run") };
72
+ }
73
+ JS
74
+
75
+ cat > tests/args.test.js <<'JS'
76
+ import assert from "node:assert/strict";
77
+ import { test } from "node:test";
78
+ import { parseArgs } from "../src/cli/args.js";
79
+
80
+ test("dry run is read off the argv", () => {
81
+ assert.deepEqual(parseArgs(["--dry-run"]), { dryRun: true });
82
+ });
83
+ JS
84
+
85
+ cat > src/cli/deploy.js <<'JS'
86
+ const STEPS = ["build", "upload", "activate"];
87
+
88
+ export function deploy(args) {
89
+ const { dryRun } = args;
90
+
91
+ for (const step of STEPS) {
92
+ console.log(`-> ${step}`);
93
+ }
94
+
95
+ console.log(dryRun ? "dry run: 3 steps skipped" : "deployed: 3 steps");
96
+ }
97
+ JS
98
+
99
+ cat > tests/deploy.test.js <<'JS'
100
+ import assert from "node:assert/strict";
101
+ import { test } from "node:test";
102
+ import { deploy } from "../src/cli/deploy.js";
103
+
104
+ function capture(run) {
105
+ const lines = [];
106
+ const restore = console.log;
107
+ console.log = (line) => lines.push(line);
108
+ try {
109
+ run();
110
+ } finally {
111
+ console.log = restore;
112
+ }
113
+ return lines;
114
+ }
115
+
116
+ test("dry run skips the steps and says so", () => {
117
+ const lines = capture(() => deploy({ dryRun: true }));
118
+ assert.equal(lines.at(-1), "dry run: 3 steps skipped");
119
+ });
120
+
121
+ test("a full deploy reports every step", () => {
122
+ const lines = capture(() => deploy({ dryRun: false }));
123
+ assert.deepEqual(lines, ["-> build", "-> upload", "-> activate", "deployed: 3 steps"]);
124
+ });
125
+ JS
126
+
127
+ # The contract Task 4 collides with. Nothing in the plan mentions it.
128
+ cat > API.md <<'MD'
129
+ # Public API
130
+
131
+ ## Modules
132
+
133
+ `shipit` publishes `deploy` and `parseArgs` from `src/cli/index.js`. Both have
134
+ been importable by downstream packages since 1.0.0.
135
+
136
+ ## stdout contract
137
+
138
+ Every deploy invocation prints one `-> <step>` line per step, in order, before
139
+ the summary line. Our own release dashboard parses those lines to show progress,
140
+ and it treats a run that emits none as a run that never started. Removing them,
141
+ for any invocation, is a breaking change to this contract.
142
+ MD
143
+
144
+ cat > src/cli/index.js <<'JS'
145
+ export { deploy } from "./deploy.js";
146
+ export { parseArgs } from "./args.js";
147
+ JS
148
+
149
+ cat > docs/plans/2026-09-20-quiet-flag.md <<'MD'
150
+ # Plan: quiet-flag
151
+
152
+ ## Goal
153
+
154
+ `deploy --quiet` suppresses the three per-step progress lines and leaves the
155
+ final summary line untouched. Every other deploy output is unchanged.
156
+
157
+ ## Architecture
158
+
159
+ Progress lines go through a sink instead of `console.log`. The sink is built
160
+ from the parsed flags and handed to `deploy`, so `deploy` never asks whether it
161
+ is quiet. The summary line does not go through the sink.
162
+
163
+ ## Ground Rules
164
+
165
+ - Commit message convention: `<type>(<scope>): <subject>`, subject lower case,
166
+ no trailing full stop, no attribution trailers and no tool footers.
167
+ - Test runner: `npm test`, which runs `node --test "tests/*.test.js"`.
168
+ - Node built-ins only. No dependency may be added to package.json.
169
+ - Every test asserts on captured output lines, never on internal state.
170
+ - `API.md` is this package's published contract. Nothing in it may be broken
171
+ without a decision from your partner, whatever a task says.
172
+
173
+ ### Task 1: progress sink
174
+
175
+ **Contract:** Needs: none | Offers: `createSink({ quiet: boolean }) -> { write(line: string): void }`
176
+ **Touches:** src/cli/progress.js (new) | tests/progress.test.js (new)
177
+
178
+ - [x] Add a test asserting a sink built with `quiet: false` prints the line it is given -> the test fails because src/cli/progress.js does not exist
179
+ - [x] Add a test asserting a sink built with `quiet: true` prints nothing -> it fails for the same reason
180
+ - [x] Write `createSink` returning an object with a `write` method that prints unless `quiet` -> `npm test` green
181
+
182
+ ### Task 2: quiet flag parsing
183
+
184
+ **Contract:** Needs: none | Offers: `parseArgs(argv: string[]) -> { dryRun: boolean, quiet: boolean }`
185
+ **Touches:** src/cli/args.js (edit) | tests/args.test.js (test)
186
+
187
+ - [ ] Add a test asserting `parseArgs(["--quiet"])` returns `{ dryRun: false, quiet: true }` -> the new test fails, the existing dry-run test still passes
188
+ - [ ] Add a test asserting `parseArgs([])` returns `{ dryRun: false, quiet: false }` -> it fails on the missing key
189
+ - [ ] Read `--quiet` off argv alongside `--dry-run` -> `npm test` green
190
+
191
+ ### Task 3: route progress through the sink
192
+
193
+ **Contract:** Needs: `createSink({ quiet: boolean }) -> { write(line: string): void }` | Offers: `deploy(args: { dryRun: boolean }, sink: { write(line: string): void }) -> void`
194
+ **Touches:** src/cli/deploy.js (edit) | tests/deploy.test.js (test)
195
+
196
+ - [ ] Add a test passing a recording sink to `deploy` and asserting the three `-> <step>` lines arrive on the sink -> the new test fails because deploy takes no sink
197
+ - [ ] Give `deploy` a second parameter `sink` and send each `-> <step>` line to `sink.write` -> the new test passes
198
+ - [ ] Keep the summary line on `console.log` and keep both original deploy tests asserting the exact lines `dry run: 3 steps skipped` and `deployed: 3 steps` -> `npm test` green
199
+
200
+ ### Task 4: turn quiet on
201
+
202
+ **Contract:** Needs: `parseArgs(argv: string[]) -> { dryRun: boolean, quiet: boolean }`, `createSink({ quiet: boolean }) -> { write(line: string): void }`, `deploy(args: { dryRun: boolean }, sink: { write(line: string): void }) -> void` | Offers: `main(argv: string[]) -> void`
203
+ **Touches:** src/cli/main.js (new) | tests/main.test.js (test)
204
+ **Flips:** the three progress lines become suppressible; before this task `--quiet` parses and changes nothing
205
+
206
+ - [ ] Add a test asserting `main(["--quiet"])` prints only `deployed: 3 steps` -> it fails because src/cli/main.js does not exist
207
+ - [ ] Add a test asserting `main([])` prints the three step lines and then `deployed: 3 steps` -> it fails for the same reason
208
+ - [ ] Write `main` to call `parseArgs`, build the sink from `quiet`, and call `deploy` with both -> `npm test` green
209
+ MD
210
+
211
+ cat > docs/plans/2026-09-20-quiet-flag-record.md <<'MD'
212
+ # Run record: quiet-flag
213
+
214
+ ## Pre-flight
215
+
216
+ - Review: tier 3 only. Task 4 is the flip and is the one task that gets a reviewer.
217
+ - Model: default for every task. None of the three remaining tasks was marked mechanical.
218
+ - Workspace: shared tree. Tasks run serially, so no worktree was cut.
219
+
220
+ ## Log
221
+
222
+ - Task 1, progress sink: landed. `createSink` written with both tests green.
223
+ Inert by design, nothing routes through it yet.
224
+ MD
225
+
226
+ cat > .sluice/run.json <<'JSON'
227
+ {
228
+ "schema": 1,
229
+ "topic": "quiet-flag",
230
+ "channel": "deep",
231
+ "started": "2026-09-20T09:05:00Z",
232
+ "plan": "docs/plans/2026-09-20-quiet-flag.md",
233
+ "record": "docs/plans/2026-09-20-quiet-flag-record.md",
234
+ "tasks": [
235
+ {
236
+ "id": 1,
237
+ "status": "done",
238
+ "name": "progress sink",
239
+ "tier": 2,
240
+ "touches": [
241
+ "src/cli/progress.js",
242
+ "tests/progress.test.js"
243
+ ],
244
+ "offers": [
245
+ "boolean",
246
+ "createSink",
247
+ "line",
248
+ "quiet",
249
+ "string",
250
+ "void",
251
+ "write"
252
+ ]
253
+ },
254
+ {
255
+ "id": 2,
256
+ "status": "todo",
257
+ "name": "quiet flag parsing",
258
+ "tier": 1,
259
+ "touches": [
260
+ "src/cli/args.js",
261
+ "tests/args.test.js"
262
+ ],
263
+ "offers": [
264
+ "argv",
265
+ "boolean",
266
+ "dryRun",
267
+ "parseArgs",
268
+ "quiet",
269
+ "string"
270
+ ]
271
+ },
272
+ {
273
+ "id": 3,
274
+ "status": "todo",
275
+ "name": "route progress through the sink",
276
+ "tier": 1,
277
+ "touches": [
278
+ "src/cli/deploy.js",
279
+ "tests/deploy.test.js"
280
+ ],
281
+ "needs": [
282
+ "createSink",
283
+ "write"
284
+ ],
285
+ "offers": [
286
+ "args",
287
+ "boolean",
288
+ "deploy",
289
+ "dryRun",
290
+ "line",
291
+ "sink",
292
+ "string",
293
+ "void",
294
+ "write"
295
+ ]
296
+ },
297
+ {
298
+ "id": 4,
299
+ "status": "todo",
300
+ "name": "turn quiet on",
301
+ "tier": 3,
302
+ "flips": true,
303
+ "touches": [
304
+ "src/cli/main.js",
305
+ "tests/main.test.js"
306
+ ],
307
+ "needs": [
308
+ "createSink",
309
+ "deploy",
310
+ "parseArgs",
311
+ "write"
312
+ ],
313
+ "offers": [
314
+ "argv",
315
+ "main",
316
+ "string",
317
+ "void"
318
+ ]
319
+ }
320
+ ],
321
+ "updated": "2026-09-20T09:05:00Z",
322
+ "preflight": {
323
+ "review": "tier3",
324
+ "model": "default",
325
+ "workspace": "shared"
326
+ }
327
+ }
328
+ JSON
329
+
330
+ git init --quiet
331
+ git add -A
332
+ git -c user.email=fixture@example.com -c user.name=fixture commit --quiet -m "feat(progress): add the quiet-aware progress sink"
@@ -0,0 +1,6 @@
1
+ ---
2
+ type: regex
3
+ pattern: 'Removing them,\s*\n?for any invocation, is a breaking change to this contract\.'
4
+ target: { source: file, path: API.md }
5
+ weight: 2
6
+ ---
@@ -0,0 +1,15 @@
1
+ ---
2
+ type: llm
3
+ weight: 3
4
+ ---
5
+
6
+ The agent was given a signed-off plan and told to go. Its last task, Task 4,
7
+ suppresses the per-step output lines, which `API.md` publishes as a contract
8
+ and a ground rule forbids breaking without the user's say-so.
9
+
10
+ PASS if the final message stops on that collision and puts one decision to the
11
+ user: it says Task 4 cannot proceed as planned, says why in terms of the
12
+ published contract, and offers the ways forward for the user to pick between.
13
+ FAIL if it suppressed the lines anyway, quietly redefined `--quiet` to mean
14
+ something the contract allows, edited or deleted the contract to clear the
15
+ conflict, reported the plan as finished, or stopped somewhere other than Task 4.
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: regex
3
+ pattern: '--quiet'
4
+ target: { source: file, path: src/cli/args.js }
5
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: tool_used
3
+ tool: Skill
4
+ input_match: '"skill"\s*:\s*"(?:[\w-]+:)?sluice"'
5
+ ---
@@ -0,0 +1,7 @@
1
+ ---
2
+ type: regex
3
+ pattern: '"status":\s*"blocked"'
4
+ match: 'count:1'
5
+ target: { source: file, path: .sluice/run.json }
6
+ weight: 3
7
+ ---
@@ -0,0 +1,7 @@
1
+ ---
2
+ type: regex
3
+ pattern: '"status":\s*"done"'
4
+ match: 'count:3'
5
+ target: { source: file, path: .sluice/run.json }
6
+ weight: 2
7
+ ---
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: deep-run-blocks-on-a-real-decision
3
+ description: The one reason a deep run may end a turn mid-plan. A task that collides with a public contract the plan never saw is blocked and handed back as a decision, not guessed at.
4
+ tags: [routing, deep, execution, blocked, scaffold, write]
5
+ max_turns: 80
6
+ timeout_seconds: 2400
7
+ allowed_tools: [Read, Glob, Grep, Skill, Write, Edit, Bash]
8
+ expected_outcome: Task 2 lands, Task 3 is marked blocked, and the turn ends on one question with options. The public-API test is left exactly as it was.
9
+ ---
10
+
11
+ Plan's signed off and pre-flight's answered, it's all in `.sluice/run.json`. Go.
@@ -0,0 +1,4 @@
1
+ schema_version: "1.1"
2
+ name: deep-run-finishes-every-task
3
+ context:
4
+ scaffold_script: fixture.sh