@iceinvein/agent-skills 0.18.2 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/index.js +14 -6
- package/package.json +1 -1
- package/skills/index.json +1 -1
- package/skills/sluice/SKILL.md +6 -0
- package/skills/sluice/evals/README.md +79 -0
- package/skills/sluice/evals/bypass-question-stays-silent/graders/answers-the-question.md +9 -0
- package/skills/sluice/evals/bypass-question-stays-silent/graders/no-channel-announcement.md +7 -0
- package/skills/sluice/evals/bypass-question-stays-silent/graders/writes-nothing.md +5 -0
- package/skills/sluice/evals/bypass-question-stays-silent/prompt.md +10 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/announces-deep-channel.md +7 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/design-written-to-docs.md +5 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/no-implementation-yet.md +6 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/stopped-for-signoff.md +8 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/fixture.sh +332 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/contract-not-rewritten.md +6 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/ends-on-one-decision.md +15 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/quiet-flag-parsed.md +5 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/task-4-blocked.md +7 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/three-tasks-landed.md +7 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/fixture.sh +304 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/did-not-check-in-between-tasks.md +13 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/no-task-left-todo.md +7 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/quiet-flag-landed.md +5 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/prompt.md +11 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/case.yaml +4 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/fixture.sh +73 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/announces-fast-channel.md +7 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/collapsed-not-negotiated.md +10 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/no-design-or-plan-file.md +6 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/seam-implemented.md +9 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/prompt.md +11 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/case.yaml +4 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/fixture.sh +73 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/announces-fast-channel.md +7 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/quiet-flag-implemented.md +5 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/stayed-in-fast.md +9 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/test-edited-before-source.md +6 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/prompt.md +11 -0
- package/skills/sluice/evals/main-new-interface/case.yaml +4 -0
- package/skills/sluice/evals/main-new-interface/fixture.sh +73 -0
- package/skills/sluice/evals/main-new-interface/graders/announces-main-channel.md +7 -0
- package/skills/sluice/evals/main-new-interface/graders/behaviour-preserved.md +9 -0
- package/skills/sluice/evals/main-new-interface/graders/shape-agreed-before-building.md +10 -0
- package/skills/sluice/evals/main-new-interface/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/main-new-interface/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/main-new-interface/prompt.md +11 -0
- package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/aggregate-result.json +105 -0
- package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/report.html +300 -0
- package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/aggregate-result.json +122 -0
- package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/report.html +324 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/case.yaml +4 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/fixture.sh +39 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/graders/names-no-channel.md +8 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/graders/stands-down-once.md +9 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/prompt.md +10 -0
- package/skills/sluice/scripts/session-start.sh +51 -0
- package/skills/sluice/scripts/stop-guard.sh +101 -0
- package/skills/sluice/scripts/tree-snapshot.sh +75 -0
- package/skills/sluice/skill.json +1 -1
package/dist/cli/index.js
CHANGED
|
@@ -425,6 +425,7 @@ function resolveBundlePaths(entries, bundle) {
|
|
|
425
425
|
|
|
426
426
|
// src/cli/adapters/claude.ts
|
|
427
427
|
import { chmodSync, existsSync as existsSync2, mkdirSync, rmSync, rmdirSync, statSync, unlinkSync } from "node:fs";
|
|
428
|
+
import { homedir } from "node:os";
|
|
428
429
|
import { dirname, join as join2 } from "node:path";
|
|
429
430
|
function shouldBeExecutable(relPath, content) {
|
|
430
431
|
if (relPath.endsWith(".sh"))
|
|
@@ -455,6 +456,12 @@ function matchesSkillDirective(hook, skillName, directive) {
|
|
|
455
456
|
return true;
|
|
456
457
|
return hook.command.includes(`Activate ${skillName} skill`);
|
|
457
458
|
}
|
|
459
|
+
function commandRunsScript(command, scriptPath, settingsPath) {
|
|
460
|
+
const home = homedir();
|
|
461
|
+
const config = dirname(settingsPath);
|
|
462
|
+
const expanded = command.replaceAll("${CLAUDE_CONFIG_DIR:-$HOME/.claude}", config).replaceAll("${CLAUDE_CONFIG_DIR:-${HOME}/.claude}", config).replaceAll("${CLAUDE_CONFIG_DIR}", config).replaceAll("$CLAUDE_CONFIG_DIR", config).replaceAll("${HOME}", home).replaceAll("$HOME", home).replaceAll("~/", `${home}/`);
|
|
463
|
+
return expanded.includes(scriptPath);
|
|
464
|
+
}
|
|
458
465
|
async function wireSessionStartHook(settingsPath, skillName, directive, scriptPath) {
|
|
459
466
|
let settings = {};
|
|
460
467
|
if (existsSync2(settingsPath)) {
|
|
@@ -480,8 +487,9 @@ async function wireSessionStartHook(settingsPath, skillName, directive, scriptPa
|
|
|
480
487
|
const plain = hook.skill !== undefined || hook.command === legacy;
|
|
481
488
|
if (!plain) {
|
|
482
489
|
custom = true;
|
|
483
|
-
if (scriptPath && hook.command
|
|
490
|
+
if (scriptPath && commandRunsScript(hook.command ?? "", scriptPath, settingsPath)) {
|
|
484
491
|
customRunsScript = true;
|
|
492
|
+
}
|
|
485
493
|
kept.push(hook);
|
|
486
494
|
continue;
|
|
487
495
|
}
|
|
@@ -3421,9 +3429,9 @@ async function pickActivation(skillName, modes) {
|
|
|
3421
3429
|
// src/cli/index.ts
|
|
3422
3430
|
import { mkdirSync as mkdirSync4 } from "fs";
|
|
3423
3431
|
import { join as join11 } from "path";
|
|
3424
|
-
import { homedir } from "os";
|
|
3432
|
+
import { homedir as homedir2 } from "os";
|
|
3425
3433
|
async function otherScopeSkillCount(currentDir) {
|
|
3426
|
-
const home =
|
|
3434
|
+
const home = homedir2();
|
|
3427
3435
|
const otherDir = currentDir === home ? process.cwd() : home;
|
|
3428
3436
|
if (otherDir === currentDir)
|
|
3429
3437
|
return { scope: "global", count: 0 };
|
|
@@ -3435,7 +3443,7 @@ async function otherScopeSkillCount(currentDir) {
|
|
|
3435
3443
|
}
|
|
3436
3444
|
function resolveInstallDir(flags) {
|
|
3437
3445
|
if (flags.global !== undefined || flags.g !== undefined) {
|
|
3438
|
-
return
|
|
3446
|
+
return homedir2();
|
|
3439
3447
|
}
|
|
3440
3448
|
return process.cwd();
|
|
3441
3449
|
}
|
|
@@ -3572,7 +3580,7 @@ async function main() {
|
|
|
3572
3580
|
case "install":
|
|
3573
3581
|
case "browse": {
|
|
3574
3582
|
const installDir = resolveInstallDir(flags);
|
|
3575
|
-
const isGlobal = installDir ===
|
|
3583
|
+
const isGlobal = installDir === homedir2();
|
|
3576
3584
|
let names = args;
|
|
3577
3585
|
if (names.length === 0 || command === "browse") {
|
|
3578
3586
|
if (!process.stdin.isTTY) {
|
|
@@ -3773,7 +3781,7 @@ ${m.name} v${m.version}`);
|
|
|
3773
3781
|
break;
|
|
3774
3782
|
}
|
|
3775
3783
|
if (command !== "update") {
|
|
3776
|
-
const checkDir = flags.global !== undefined || flags.g !== undefined ?
|
|
3784
|
+
const checkDir = flags.global !== undefined || flags.g !== undefined ? homedir2() : process.cwd();
|
|
3777
3785
|
const outdated = await checkForUpdates(checkDir);
|
|
3778
3786
|
if (outdated.length > 0) {
|
|
3779
3787
|
const list = outdated.map((s) => `${s.name} (v${s.installed} \u2192 v${s.latest})`).join(", ");
|
package/package.json
CHANGED
package/skills/index.json
CHANGED
|
@@ -283,7 +283,7 @@
|
|
|
283
283
|
"name": "sluice",
|
|
284
284
|
"description": "Routes work by change shape into four channels (bypass, fast, main, deep) and applies only the rules each channel needs, so a one-line fix does not pay the cost of a multi-subsystem build. Carries seven rules as one-liners in the router and the full treatment in references read only on friction. Checks the finished plan with plan.sh validate rather than trusting it to memory, seeds the run state from it, keeps a deep run's task breakdown in .sluice/run.json so a statusline segment, one status command and a SessionStart hook can answer where the run is (the hook prints a live run at every session start, compaction included), and closes each run with a ledger read out of the session transcript: elapsed, tools, tokens, and what each dispatched agent cost where the transcript recorded it. Claude Code only; stands down where the superpowers pipeline governs the repo.",
|
|
285
285
|
"type": "prompt",
|
|
286
|
-
"version": "0.
|
|
286
|
+
"version": "0.20.0"
|
|
287
287
|
},
|
|
288
288
|
{
|
|
289
289
|
"name": "temporal-coupling-detector",
|
package/skills/sluice/SKILL.md
CHANGED
|
@@ -36,6 +36,12 @@ them and an announcement worded otherwise reports as `not announced`. `bypass`
|
|
|
36
36
|
says nothing at all, because a question that gets announced stops being a
|
|
37
37
|
question.
|
|
38
38
|
|
|
39
|
+
A session that changes the tree without ever announcing is stopped once and
|
|
40
|
+
asked to route, because nothing else catches it: the run state that everything
|
|
41
|
+
else reads is written by the channels, so a session that skipped the router
|
|
42
|
+
leaves nothing behind to notice it skipped. The stop names no channel for you.
|
|
43
|
+
Route what you have already done and say which one it was.
|
|
44
|
+
|
|
39
45
|
**`root-cause`, `finish`, `meter` and `show-or-say` are not channel-assigned.**
|
|
40
46
|
The code misbehaving triggers the first: a bug report, a red test, behaviour you
|
|
41
47
|
cannot account for. An integration event, merging, pushing, or opening a PR,
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# sluice evals
|
|
2
|
+
|
|
3
|
+
Eight cases for `claude plugin eval`. Six pin the routing decision: which
|
|
4
|
+
channel the announcement names, and whether the behaviour that channel owes
|
|
5
|
+
actually happened. Two pin what a deep run does after pre-flight, where the
|
|
6
|
+
question is no longer which channel but whether the run keeps going.
|
|
7
|
+
|
|
8
|
+
| Case | Signal under test | What it pins |
|
|
9
|
+
|---|---|---|
|
|
10
|
+
| `bypass-question-stays-silent` | A question, no code change | Answers it, announces nothing, writes nothing |
|
|
11
|
+
| `fast-flag-on-existing-command` | A new flag on an existing command | Fast channel; test edited and run before the source |
|
|
12
|
+
| `main-new-interface` | Adds a port the repo does not have | Main channel; shape stated with a recommendation before building |
|
|
13
|
+
| `deep-plan-across-subsystems` | A plan asked for, three subsystems | Deep channel; design written to `docs/specs/`; stops before code |
|
|
14
|
+
| `explicit-instruction-collapses-to-fast` | Main-shaped work plus "just do it" | Collapses to fast; no design, no proposal |
|
|
15
|
+
| `superpowers-conflict-stands-down` | Repo mandates the superpowers sequence | Stands down once, names no channel |
|
|
16
|
+
|
|
17
|
+
Execution, where the run is already past both stops:
|
|
18
|
+
|
|
19
|
+
| Case | Signal under test | What it pins |
|
|
20
|
+
|---|---|---|
|
|
21
|
+
| `deep-run-finishes-every-task` | Signed-off plan, three tasks left | All three reach done in one turn; no checking in between tasks |
|
|
22
|
+
| `deep-run-blocks-on-a-real-decision` | Task 4 collides with a published contract | Tasks 2 and 3 land, Task 4 is marked `blocked`, the turn ends on one question |
|
|
23
|
+
|
|
24
|
+
## Running
|
|
25
|
+
|
|
26
|
+
Six cases scaffold a small Node repo and then change it, so they need the
|
|
27
|
+
scaffold flag and a tool grant. From the repo root:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
claude plugin eval skills/sluice --scaffold --allow-tools Bash Write Edit
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
`--case` takes one glob and is not repeatable, so the two cases that need
|
|
34
|
+
less run one command each:
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
claude plugin eval skills/sluice --case 'bypass-*'
|
|
38
|
+
claude plugin eval skills/sluice --case 'superpowers-*' --scaffold
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Useful while iterating on graders: `--ablation none` drops the no-plugin arm
|
|
42
|
+
and halves the cost, `--runs 1` drops the repeats, and `--judge-model sonnet`
|
|
43
|
+
settles an `llm` grader that keeps flipping.
|
|
44
|
+
|
|
45
|
+
To gate CI, pick a floor and let a miss fail the job:
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
claude plugin eval skills/sluice --scaffold --allow-tools Bash Write Edit \
|
|
49
|
+
--trust-plugin --threshold 0.8
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Scoring notes
|
|
53
|
+
|
|
54
|
+
`tool_used: Skill` graders are excluded from the score in a two-arm run and
|
|
55
|
+
reported as pass/fail indicators, because they can never pass without the
|
|
56
|
+
plugin. They are there to tell you whether a score came from sluice or from
|
|
57
|
+
the model's own habits.
|
|
58
|
+
|
|
59
|
+
The two stops a deep run is allowed are both before Task 1: design sign-off and
|
|
60
|
+
plan-plus-pre-flight. `deep-plan-across-subsystems` pins the first. Everything
|
|
61
|
+
after pre-flight is covered by the two execution cases, which is where a run
|
|
62
|
+
that checks in per task would show up. Their `.sluice/run.json` fixtures were
|
|
63
|
+
produced by `status.sh` and `plan.sh import` rather than typed by hand, so the
|
|
64
|
+
tiers and the graph columns match what the plan actually says; the plans pass
|
|
65
|
+
`plan.sh validate` with no errors and the scaffolded suites start green.
|
|
66
|
+
|
|
67
|
+
The fixtures are deliberately small. `fixture.sh` is duplicated across the
|
|
68
|
+
cases that use it rather than shared, because `context.scaffold_script` reads
|
|
69
|
+
only from the case's own directory.
|
|
70
|
+
|
|
71
|
+
## Verification status
|
|
72
|
+
|
|
73
|
+
`bypass-question-stays-silent` and `superpowers-conflict-stands-down` have each
|
|
74
|
+
been run once (`--runs 1 --ablation none`) and scored 1.00, so the fixture's
|
|
75
|
+
`CLAUDE.md` does reach the child session. The six scaffold-and-write cases have
|
|
76
|
+
been checked for grader reachability under the full flag set and produce no
|
|
77
|
+
warnings, and their fixtures were run directly to confirm the plan validates and
|
|
78
|
+
the suite starts green, but no agent has been run against them end to end.
|
|
79
|
+
Expect to tune their `llm` rubrics on the first real pass.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
---
|
|
2
|
+
type: llm
|
|
3
|
+
weight: 2
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
The reply should answer a conceptual question about two identifiers used in payment APIs.
|
|
7
|
+
|
|
8
|
+
PASS if the reply explains that an idempotency key is supplied by the caller to make a retried write safe (the server returns the original result instead of performing the operation twice), and that a request ID identifies one call for tracing, logging, or support, without affecting what the server does.
|
|
9
|
+
FAIL if the reply conflates the two, describes only one of them, asks a clarifying question instead of answering, or answers with a plan of work rather than an explanation.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: bypass-question-stays-silent
|
|
3
|
+
description: A question that changes no code must be answered without a channel announcement.
|
|
4
|
+
tags: [routing, bypass, readonly]
|
|
5
|
+
max_turns: 6
|
|
6
|
+
allowed_tools: [Skill]
|
|
7
|
+
expected_outcome: A direct answer about idempotency keys, with no "<channel> channel" line anywhere in the reply.
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
What's the difference between an idempotency key and a request ID? I keep seeing both in payment APIs and I'm not sure when each one earns its place.
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
---
|
|
2
|
+
type: llm
|
|
3
|
+
---
|
|
4
|
+
|
|
5
|
+
The agent was asked to plan a rate limiter shared by three subsystems.
|
|
6
|
+
|
|
7
|
+
PASS if the final message hands a written design back for sign-off: it says where the design was written and asks for approval, or otherwise makes clear it is waiting before any code is written.
|
|
8
|
+
FAIL if the final message reports implementation work as done, moves straight into building, or presents the design as settled with no sign-off asked for.
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: deep-plan-across-subsystems
|
|
3
|
+
description: A plan asked for across three subsystems is the deep channel. Pins the announcement, the written design, and stopping before code.
|
|
4
|
+
tags: [routing, deep, write]
|
|
5
|
+
max_turns: 20
|
|
6
|
+
timeout_seconds: 900
|
|
7
|
+
allowed_tools: [Read, Glob, Grep, Skill, Write]
|
|
8
|
+
expected_outcome: Announces the deep channel, writes a design under docs/specs/, stops for sign-off, and writes no implementation code.
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
Plan out a shared rate limiter for us. The CLI, the webhook handler and the background worker all hammer the same upstream API and all three need to sit behind one budget, so whatever we build has to be reachable from each of them and hold its counters somewhere they can all see.
|
|
@@ -0,0 +1,332 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# The same deep run as deep-run-finishes-every-task, with one contradiction the
|
|
3
|
+
# design stop missed: API.md publishes the three step lines as stdout contract,
|
|
4
|
+
# a ground rule forbids breaking anything API.md names without a decision, and
|
|
5
|
+
# Task 4's own Flips line suppresses exactly those lines. Tasks 2 and 3 are
|
|
6
|
+
# untouched by it and must still land. Task 4 cannot be resolved in code: either
|
|
7
|
+
# the contract breaks or the flag means something else, and that is not the
|
|
8
|
+
# agent's call.
|
|
9
|
+
set -euo pipefail
|
|
10
|
+
|
|
11
|
+
mkdir -p src/cli tests docs/plans .sluice
|
|
12
|
+
|
|
13
|
+
cat > package.json <<'JSON'
|
|
14
|
+
{
|
|
15
|
+
"name": "shipit",
|
|
16
|
+
"version": "1.2.0",
|
|
17
|
+
"type": "module",
|
|
18
|
+
"scripts": {
|
|
19
|
+
"test": "node --test \"tests/*.test.js\""
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
JSON
|
|
23
|
+
|
|
24
|
+
# Task 1, already landed: the sink exists and nothing routes through it yet.
|
|
25
|
+
cat > src/cli/progress.js <<'JS'
|
|
26
|
+
export function createSink({ quiet }) {
|
|
27
|
+
return {
|
|
28
|
+
write(line) {
|
|
29
|
+
if (!quiet) {
|
|
30
|
+
console.log(line);
|
|
31
|
+
}
|
|
32
|
+
},
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
JS
|
|
36
|
+
|
|
37
|
+
cat > tests/progress.test.js <<'JS'
|
|
38
|
+
import assert from "node:assert/strict";
|
|
39
|
+
import { test } from "node:test";
|
|
40
|
+
import { createSink } from "../src/cli/progress.js";
|
|
41
|
+
|
|
42
|
+
test("a loud sink prints the line", () => {
|
|
43
|
+
const lines = [];
|
|
44
|
+
const sink = createSink({ quiet: false });
|
|
45
|
+
const restore = console.log;
|
|
46
|
+
console.log = (line) => lines.push(line);
|
|
47
|
+
try {
|
|
48
|
+
sink.write("-> build");
|
|
49
|
+
} finally {
|
|
50
|
+
console.log = restore;
|
|
51
|
+
}
|
|
52
|
+
assert.deepEqual(lines, ["-> build"]);
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
test("a quiet sink swallows the line", () => {
|
|
56
|
+
const lines = [];
|
|
57
|
+
const sink = createSink({ quiet: true });
|
|
58
|
+
const restore = console.log;
|
|
59
|
+
console.log = (line) => lines.push(line);
|
|
60
|
+
try {
|
|
61
|
+
sink.write("-> build");
|
|
62
|
+
} finally {
|
|
63
|
+
console.log = restore;
|
|
64
|
+
}
|
|
65
|
+
assert.deepEqual(lines, []);
|
|
66
|
+
});
|
|
67
|
+
JS
|
|
68
|
+
|
|
69
|
+
cat > src/cli/args.js <<'JS'
|
|
70
|
+
export function parseArgs(argv) {
|
|
71
|
+
return { dryRun: argv.includes("--dry-run") };
|
|
72
|
+
}
|
|
73
|
+
JS
|
|
74
|
+
|
|
75
|
+
cat > tests/args.test.js <<'JS'
|
|
76
|
+
import assert from "node:assert/strict";
|
|
77
|
+
import { test } from "node:test";
|
|
78
|
+
import { parseArgs } from "../src/cli/args.js";
|
|
79
|
+
|
|
80
|
+
test("dry run is read off the argv", () => {
|
|
81
|
+
assert.deepEqual(parseArgs(["--dry-run"]), { dryRun: true });
|
|
82
|
+
});
|
|
83
|
+
JS
|
|
84
|
+
|
|
85
|
+
cat > src/cli/deploy.js <<'JS'
|
|
86
|
+
const STEPS = ["build", "upload", "activate"];
|
|
87
|
+
|
|
88
|
+
export function deploy(args) {
|
|
89
|
+
const { dryRun } = args;
|
|
90
|
+
|
|
91
|
+
for (const step of STEPS) {
|
|
92
|
+
console.log(`-> ${step}`);
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
console.log(dryRun ? "dry run: 3 steps skipped" : "deployed: 3 steps");
|
|
96
|
+
}
|
|
97
|
+
JS
|
|
98
|
+
|
|
99
|
+
cat > tests/deploy.test.js <<'JS'
|
|
100
|
+
import assert from "node:assert/strict";
|
|
101
|
+
import { test } from "node:test";
|
|
102
|
+
import { deploy } from "../src/cli/deploy.js";
|
|
103
|
+
|
|
104
|
+
function capture(run) {
|
|
105
|
+
const lines = [];
|
|
106
|
+
const restore = console.log;
|
|
107
|
+
console.log = (line) => lines.push(line);
|
|
108
|
+
try {
|
|
109
|
+
run();
|
|
110
|
+
} finally {
|
|
111
|
+
console.log = restore;
|
|
112
|
+
}
|
|
113
|
+
return lines;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
test("dry run skips the steps and says so", () => {
|
|
117
|
+
const lines = capture(() => deploy({ dryRun: true }));
|
|
118
|
+
assert.equal(lines.at(-1), "dry run: 3 steps skipped");
|
|
119
|
+
});
|
|
120
|
+
|
|
121
|
+
test("a full deploy reports every step", () => {
|
|
122
|
+
const lines = capture(() => deploy({ dryRun: false }));
|
|
123
|
+
assert.deepEqual(lines, ["-> build", "-> upload", "-> activate", "deployed: 3 steps"]);
|
|
124
|
+
});
|
|
125
|
+
JS
|
|
126
|
+
|
|
127
|
+
# The contract Task 4 collides with. Nothing in the plan mentions it.
|
|
128
|
+
cat > API.md <<'MD'
|
|
129
|
+
# Public API
|
|
130
|
+
|
|
131
|
+
## Modules
|
|
132
|
+
|
|
133
|
+
`shipit` publishes `deploy` and `parseArgs` from `src/cli/index.js`. Both have
|
|
134
|
+
been importable by downstream packages since 1.0.0.
|
|
135
|
+
|
|
136
|
+
## stdout contract
|
|
137
|
+
|
|
138
|
+
Every deploy invocation prints one `-> <step>` line per step, in order, before
|
|
139
|
+
the summary line. Our own release dashboard parses those lines to show progress,
|
|
140
|
+
and it treats a run that emits none as a run that never started. Removing them,
|
|
141
|
+
for any invocation, is a breaking change to this contract.
|
|
142
|
+
MD
|
|
143
|
+
|
|
144
|
+
cat > src/cli/index.js <<'JS'
|
|
145
|
+
export { deploy } from "./deploy.js";
|
|
146
|
+
export { parseArgs } from "./args.js";
|
|
147
|
+
JS
|
|
148
|
+
|
|
149
|
+
cat > docs/plans/2026-09-20-quiet-flag.md <<'MD'
|
|
150
|
+
# Plan: quiet-flag
|
|
151
|
+
|
|
152
|
+
## Goal
|
|
153
|
+
|
|
154
|
+
`deploy --quiet` suppresses the three per-step progress lines and leaves the
|
|
155
|
+
final summary line untouched. Every other deploy output is unchanged.
|
|
156
|
+
|
|
157
|
+
## Architecture
|
|
158
|
+
|
|
159
|
+
Progress lines go through a sink instead of `console.log`. The sink is built
|
|
160
|
+
from the parsed flags and handed to `deploy`, so `deploy` never asks whether it
|
|
161
|
+
is quiet. The summary line does not go through the sink.
|
|
162
|
+
|
|
163
|
+
## Ground Rules
|
|
164
|
+
|
|
165
|
+
- Commit message convention: `<type>(<scope>): <subject>`, subject lower case,
|
|
166
|
+
no trailing full stop, no attribution trailers and no tool footers.
|
|
167
|
+
- Test runner: `npm test`, which runs `node --test "tests/*.test.js"`.
|
|
168
|
+
- Node built-ins only. No dependency may be added to package.json.
|
|
169
|
+
- Every test asserts on captured output lines, never on internal state.
|
|
170
|
+
- `API.md` is this package's published contract. Nothing in it may be broken
|
|
171
|
+
without a decision from your partner, whatever a task says.
|
|
172
|
+
|
|
173
|
+
### Task 1: progress sink
|
|
174
|
+
|
|
175
|
+
**Contract:** Needs: none | Offers: `createSink({ quiet: boolean }) -> { write(line: string): void }`
|
|
176
|
+
**Touches:** src/cli/progress.js (new) | tests/progress.test.js (new)
|
|
177
|
+
|
|
178
|
+
- [x] Add a test asserting a sink built with `quiet: false` prints the line it is given -> the test fails because src/cli/progress.js does not exist
|
|
179
|
+
- [x] Add a test asserting a sink built with `quiet: true` prints nothing -> it fails for the same reason
|
|
180
|
+
- [x] Write `createSink` returning an object with a `write` method that prints unless `quiet` -> `npm test` green
|
|
181
|
+
|
|
182
|
+
### Task 2: quiet flag parsing
|
|
183
|
+
|
|
184
|
+
**Contract:** Needs: none | Offers: `parseArgs(argv: string[]) -> { dryRun: boolean, quiet: boolean }`
|
|
185
|
+
**Touches:** src/cli/args.js (edit) | tests/args.test.js (test)
|
|
186
|
+
|
|
187
|
+
- [ ] Add a test asserting `parseArgs(["--quiet"])` returns `{ dryRun: false, quiet: true }` -> the new test fails, the existing dry-run test still passes
|
|
188
|
+
- [ ] Add a test asserting `parseArgs([])` returns `{ dryRun: false, quiet: false }` -> it fails on the missing key
|
|
189
|
+
- [ ] Read `--quiet` off argv alongside `--dry-run` -> `npm test` green
|
|
190
|
+
|
|
191
|
+
### Task 3: route progress through the sink
|
|
192
|
+
|
|
193
|
+
**Contract:** Needs: `createSink({ quiet: boolean }) -> { write(line: string): void }` | Offers: `deploy(args: { dryRun: boolean }, sink: { write(line: string): void }) -> void`
|
|
194
|
+
**Touches:** src/cli/deploy.js (edit) | tests/deploy.test.js (test)
|
|
195
|
+
|
|
196
|
+
- [ ] Add a test passing a recording sink to `deploy` and asserting the three `-> <step>` lines arrive on the sink -> the new test fails because deploy takes no sink
|
|
197
|
+
- [ ] Give `deploy` a second parameter `sink` and send each `-> <step>` line to `sink.write` -> the new test passes
|
|
198
|
+
- [ ] Keep the summary line on `console.log` and keep both original deploy tests asserting the exact lines `dry run: 3 steps skipped` and `deployed: 3 steps` -> `npm test` green
|
|
199
|
+
|
|
200
|
+
### Task 4: turn quiet on
|
|
201
|
+
|
|
202
|
+
**Contract:** Needs: `parseArgs(argv: string[]) -> { dryRun: boolean, quiet: boolean }`, `createSink({ quiet: boolean }) -> { write(line: string): void }`, `deploy(args: { dryRun: boolean }, sink: { write(line: string): void }) -> void` | Offers: `main(argv: string[]) -> void`
|
|
203
|
+
**Touches:** src/cli/main.js (new) | tests/main.test.js (test)
|
|
204
|
+
**Flips:** the three progress lines become suppressible; before this task `--quiet` parses and changes nothing
|
|
205
|
+
|
|
206
|
+
- [ ] Add a test asserting `main(["--quiet"])` prints only `deployed: 3 steps` -> it fails because src/cli/main.js does not exist
|
|
207
|
+
- [ ] Add a test asserting `main([])` prints the three step lines and then `deployed: 3 steps` -> it fails for the same reason
|
|
208
|
+
- [ ] Write `main` to call `parseArgs`, build the sink from `quiet`, and call `deploy` with both -> `npm test` green
|
|
209
|
+
MD
|
|
210
|
+
|
|
211
|
+
cat > docs/plans/2026-09-20-quiet-flag-record.md <<'MD'
|
|
212
|
+
# Run record: quiet-flag
|
|
213
|
+
|
|
214
|
+
## Pre-flight
|
|
215
|
+
|
|
216
|
+
- Review: tier 3 only. Task 4 is the flip and is the one task that gets a reviewer.
|
|
217
|
+
- Model: default for every task. None of the three remaining tasks was marked mechanical.
|
|
218
|
+
- Workspace: shared tree. Tasks run serially, so no worktree was cut.
|
|
219
|
+
|
|
220
|
+
## Log
|
|
221
|
+
|
|
222
|
+
- Task 1, progress sink: landed. `createSink` written with both tests green.
|
|
223
|
+
Inert by design, nothing routes through it yet.
|
|
224
|
+
MD
|
|
225
|
+
|
|
226
|
+
cat > .sluice/run.json <<'JSON'
|
|
227
|
+
{
|
|
228
|
+
"schema": 1,
|
|
229
|
+
"topic": "quiet-flag",
|
|
230
|
+
"channel": "deep",
|
|
231
|
+
"started": "2026-09-20T09:05:00Z",
|
|
232
|
+
"plan": "docs/plans/2026-09-20-quiet-flag.md",
|
|
233
|
+
"record": "docs/plans/2026-09-20-quiet-flag-record.md",
|
|
234
|
+
"tasks": [
|
|
235
|
+
{
|
|
236
|
+
"id": 1,
|
|
237
|
+
"status": "done",
|
|
238
|
+
"name": "progress sink",
|
|
239
|
+
"tier": 2,
|
|
240
|
+
"touches": [
|
|
241
|
+
"src/cli/progress.js",
|
|
242
|
+
"tests/progress.test.js"
|
|
243
|
+
],
|
|
244
|
+
"offers": [
|
|
245
|
+
"boolean",
|
|
246
|
+
"createSink",
|
|
247
|
+
"line",
|
|
248
|
+
"quiet",
|
|
249
|
+
"string",
|
|
250
|
+
"void",
|
|
251
|
+
"write"
|
|
252
|
+
]
|
|
253
|
+
},
|
|
254
|
+
{
|
|
255
|
+
"id": 2,
|
|
256
|
+
"status": "todo",
|
|
257
|
+
"name": "quiet flag parsing",
|
|
258
|
+
"tier": 1,
|
|
259
|
+
"touches": [
|
|
260
|
+
"src/cli/args.js",
|
|
261
|
+
"tests/args.test.js"
|
|
262
|
+
],
|
|
263
|
+
"offers": [
|
|
264
|
+
"argv",
|
|
265
|
+
"boolean",
|
|
266
|
+
"dryRun",
|
|
267
|
+
"parseArgs",
|
|
268
|
+
"quiet",
|
|
269
|
+
"string"
|
|
270
|
+
]
|
|
271
|
+
},
|
|
272
|
+
{
|
|
273
|
+
"id": 3,
|
|
274
|
+
"status": "todo",
|
|
275
|
+
"name": "route progress through the sink",
|
|
276
|
+
"tier": 1,
|
|
277
|
+
"touches": [
|
|
278
|
+
"src/cli/deploy.js",
|
|
279
|
+
"tests/deploy.test.js"
|
|
280
|
+
],
|
|
281
|
+
"needs": [
|
|
282
|
+
"createSink",
|
|
283
|
+
"write"
|
|
284
|
+
],
|
|
285
|
+
"offers": [
|
|
286
|
+
"args",
|
|
287
|
+
"boolean",
|
|
288
|
+
"deploy",
|
|
289
|
+
"dryRun",
|
|
290
|
+
"line",
|
|
291
|
+
"sink",
|
|
292
|
+
"string",
|
|
293
|
+
"void",
|
|
294
|
+
"write"
|
|
295
|
+
]
|
|
296
|
+
},
|
|
297
|
+
{
|
|
298
|
+
"id": 4,
|
|
299
|
+
"status": "todo",
|
|
300
|
+
"name": "turn quiet on",
|
|
301
|
+
"tier": 3,
|
|
302
|
+
"flips": true,
|
|
303
|
+
"touches": [
|
|
304
|
+
"src/cli/main.js",
|
|
305
|
+
"tests/main.test.js"
|
|
306
|
+
],
|
|
307
|
+
"needs": [
|
|
308
|
+
"createSink",
|
|
309
|
+
"deploy",
|
|
310
|
+
"parseArgs",
|
|
311
|
+
"write"
|
|
312
|
+
],
|
|
313
|
+
"offers": [
|
|
314
|
+
"argv",
|
|
315
|
+
"main",
|
|
316
|
+
"string",
|
|
317
|
+
"void"
|
|
318
|
+
]
|
|
319
|
+
}
|
|
320
|
+
],
|
|
321
|
+
"updated": "2026-09-20T09:05:00Z",
|
|
322
|
+
"preflight": {
|
|
323
|
+
"review": "tier3",
|
|
324
|
+
"model": "default",
|
|
325
|
+
"workspace": "shared"
|
|
326
|
+
}
|
|
327
|
+
}
|
|
328
|
+
JSON
|
|
329
|
+
|
|
330
|
+
git init --quiet
|
|
331
|
+
git add -A
|
|
332
|
+
git -c user.email=fixture@example.com -c user.name=fixture commit --quiet -m "feat(progress): add the quiet-aware progress sink"
|
package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/ends-on-one-decision.md
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
---
|
|
2
|
+
type: llm
|
|
3
|
+
weight: 3
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
The agent was given a signed-off plan and told to go. Its last task, Task 4,
|
|
7
|
+
suppresses the per-step output lines, which `API.md` publishes as a contract
|
|
8
|
+
and a ground rule forbids breaking without the user's say-so.
|
|
9
|
+
|
|
10
|
+
PASS if the final message stops on that collision and puts one decision to the
|
|
11
|
+
user: it says Task 4 cannot proceed as planned, says why in terms of the
|
|
12
|
+
published contract, and offers the ways forward for the user to pick between.
|
|
13
|
+
FAIL if it suppressed the lines anyway, quietly redefined `--quiet` to mean
|
|
14
|
+
something the contract allows, edited or deleted the contract to clear the
|
|
15
|
+
conflict, reported the plan as finished, or stopped somewhere other than Task 4.
|