@iceinvein/agent-skills 0.18.2 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/index.js +14 -6
- package/package.json +1 -1
- package/skills/index.json +1 -1
- package/skills/sluice/SKILL.md +6 -0
- package/skills/sluice/evals/README.md +79 -0
- package/skills/sluice/evals/bypass-question-stays-silent/graders/answers-the-question.md +9 -0
- package/skills/sluice/evals/bypass-question-stays-silent/graders/no-channel-announcement.md +7 -0
- package/skills/sluice/evals/bypass-question-stays-silent/graders/writes-nothing.md +5 -0
- package/skills/sluice/evals/bypass-question-stays-silent/prompt.md +10 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/announces-deep-channel.md +7 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/design-written-to-docs.md +5 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/no-implementation-yet.md +6 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/stopped-for-signoff.md +8 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/fixture.sh +332 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/contract-not-rewritten.md +6 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/ends-on-one-decision.md +15 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/quiet-flag-parsed.md +5 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/task-4-blocked.md +7 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/graders/three-tasks-landed.md +7 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/fixture.sh +304 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/did-not-check-in-between-tasks.md +13 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/no-task-left-todo.md +7 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/quiet-flag-landed.md +5 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/prompt.md +11 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/case.yaml +4 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/fixture.sh +73 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/announces-fast-channel.md +7 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/collapsed-not-negotiated.md +10 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/no-design-or-plan-file.md +6 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/seam-implemented.md +9 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/prompt.md +11 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/case.yaml +4 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/fixture.sh +73 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/announces-fast-channel.md +7 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/quiet-flag-implemented.md +5 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/stayed-in-fast.md +9 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/test-edited-before-source.md +6 -0
- package/skills/sluice/evals/fast-flag-on-existing-command/prompt.md +11 -0
- package/skills/sluice/evals/main-new-interface/case.yaml +4 -0
- package/skills/sluice/evals/main-new-interface/fixture.sh +73 -0
- package/skills/sluice/evals/main-new-interface/graders/announces-main-channel.md +7 -0
- package/skills/sluice/evals/main-new-interface/graders/behaviour-preserved.md +9 -0
- package/skills/sluice/evals/main-new-interface/graders/shape-agreed-before-building.md +10 -0
- package/skills/sluice/evals/main-new-interface/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/main-new-interface/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/main-new-interface/prompt.md +11 -0
- package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/aggregate-result.json +105 -0
- package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/report.html +300 -0
- package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/aggregate-result.json +122 -0
- package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/report.html +324 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/case.yaml +4 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/fixture.sh +39 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/graders/names-no-channel.md +8 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/graders/stands-down-once.md +9 -0
- package/skills/sluice/evals/superpowers-conflict-stands-down/prompt.md +10 -0
- package/skills/sluice/scripts/session-start.sh +51 -0
- package/skills/sluice/scripts/stop-guard.sh +101 -0
- package/skills/sluice/scripts/tree-snapshot.sh +75 -0
- package/skills/sluice/skill.json +1 -1
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"claudeVersion": "2.1.278",
|
|
4
|
+
"startedAt": "2026-09-20T01:52:06.287Z",
|
|
5
|
+
"durationSeconds": 33,
|
|
6
|
+
"costUsd": 0.11528100000000001,
|
|
7
|
+
"partial": false,
|
|
8
|
+
"suite": {
|
|
9
|
+
"root": "/Users/dikrana/Documents/projects/agent-skills/skills/sluice",
|
|
10
|
+
"ablation": "none",
|
|
11
|
+
"caseFilter": "bypass-*",
|
|
12
|
+
"threshold": 1,
|
|
13
|
+
"concurrency": 1,
|
|
14
|
+
"plugins": [
|
|
15
|
+
{
|
|
16
|
+
"name": "sluice",
|
|
17
|
+
"path": "/Users/dikrana/Documents/projects/agent-skills/skills/sluice"
|
|
18
|
+
}
|
|
19
|
+
]
|
|
20
|
+
},
|
|
21
|
+
"cases": [
|
|
22
|
+
{
|
|
23
|
+
"name": "bypass-question-stays-silent",
|
|
24
|
+
"dir": "evals/bypass-question-stays-silent",
|
|
25
|
+
"source": "prose",
|
|
26
|
+
"promptMarkdown": "What's the difference between an idempotency key and a request ID? I keep seeing both in payment APIs and I'm not sure when each one earns its place.",
|
|
27
|
+
"runsPerCase": 3,
|
|
28
|
+
"timeoutSeconds": 300,
|
|
29
|
+
"maxTurns": 6,
|
|
30
|
+
"graders": [
|
|
31
|
+
{
|
|
32
|
+
"name": "answers-the-question",
|
|
33
|
+
"type": "llm",
|
|
34
|
+
"weight": 2,
|
|
35
|
+
"graderMarkdown": "The reply should answer a conceptual question about two identifiers used in payment APIs.\n\nPASS if the reply explains that an idempotency key is supplied by the caller to make a retried write safe (the server returns the original result instead of performing the operation twice), and that a request ID identifies one call for tracing, logging, or support, without affecting what the server does.\nFAIL if the reply conflates the two, describes only one of them, asks a clarifying question instead of answering, or answers with a plan of work rather than an explanation.",
|
|
36
|
+
"config": {
|
|
37
|
+
"criteria": "The reply should answer a conceptual question about two identifiers used in payment APIs.\n\nPASS if the reply explains that an idempotency key is supplied by the caller to make a retried write safe (the server returns the original result instead of performing the operation twice), and that a request ID identifies one call for tracing, logging, or support, without affecting what the server does.\nFAIL if the reply conflates the two, describes only one of them, asks a clarifying question instead of answering, or answers with a plan of work rather than an explanation.",
|
|
38
|
+
"focus": "last_message"
|
|
39
|
+
}
|
|
40
|
+
},
|
|
41
|
+
{
|
|
42
|
+
"name": "no-channel-announcement",
|
|
43
|
+
"type": "regex",
|
|
44
|
+
"weight": 1,
|
|
45
|
+
"config": {
|
|
46
|
+
"target": "last_message",
|
|
47
|
+
"pattern": "(bypass|fast|main|deep)\\s+channel",
|
|
48
|
+
"flags": "i",
|
|
49
|
+
"match": "not_contains"
|
|
50
|
+
}
|
|
51
|
+
},
|
|
52
|
+
{
|
|
53
|
+
"name": "writes-nothing",
|
|
54
|
+
"type": "file_exists",
|
|
55
|
+
"weight": 1,
|
|
56
|
+
"config": {
|
|
57
|
+
"path": "**/*",
|
|
58
|
+
"exists": false
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
],
|
|
62
|
+
"arms": {
|
|
63
|
+
"with": [
|
|
64
|
+
{
|
|
65
|
+
"score": 1,
|
|
66
|
+
"passed": true,
|
|
67
|
+
"turns": 1,
|
|
68
|
+
"costUsd": 0.11528100000000001,
|
|
69
|
+
"judgeCostUsd": 0.0030900000000000003,
|
|
70
|
+
"durationSeconds": 33,
|
|
71
|
+
"startedAt": "2026-09-20T01:52:06.306Z",
|
|
72
|
+
"error": null,
|
|
73
|
+
"tracePath": "/private/tmp/e-cN9KZf/out/trace.jsonl",
|
|
74
|
+
"skippedPaidGraders": false,
|
|
75
|
+
"graders": [
|
|
76
|
+
{
|
|
77
|
+
"name": "answers-the-question",
|
|
78
|
+
"passed": true,
|
|
79
|
+
"weight": 2,
|
|
80
|
+
"explanation": "judge votes: PASS PASS PASS",
|
|
81
|
+
"withOnly": false,
|
|
82
|
+
"scored": true,
|
|
83
|
+
"judgeVotes": [
|
|
84
|
+
true,
|
|
85
|
+
true,
|
|
86
|
+
true
|
|
87
|
+
],
|
|
88
|
+
"evidence": "They answer different questions, and the cleanest way to keep them straight is cardinality across a retry:\n\n**One idempotency key spans many request IDs.** If your client sends a charge, times out, and retries three times, that's one idempotency key and four request IDs.\n\n## Idempotency key\n\nClient-generated, semantic, part of the API contract. It means: *\"these attempts are the same logical operation — do the work at most once.\"*\n\nThe server has to actually do something with it: store key → (request fingerprint, response), and on a repeat return the stored response instead of re-executing. That implies real machinery:\n\n- **Conflict handling** — same key, different payload is a client bug; reject it (422/409) rather than silently replaying the old response.\n- **In-flight handling** — two concurrent requests with the same key need a lock and a 409, not two charges.\n- **Scoping** — keys must be namespaced per API key/account, or one tenant's UUID collision becomes another tenant's replayed response.\n- **Expiry** — typically 24h (Stripe's window). After that the key is forgotten and a retry executes for real, which is why it's a retry-safety mechanism, not a permanent dedup ledger.\n\nThe subtle client-side requirement: generate the key *before* the first attempt and persist it alongside the pending operation. A key that only lives in the retry loop's memory is lost on process restart, and the retry-after-crash — the exact case you most needed it for — double-charges.\n\n## Request ID\n\nPer-attempt identifier, usually server-generated and echoed in a header (`X-Request-Id`) and in error bodies. It's for correlation: logs, traces, support tickets, \"here's the ID, what happened to my call.\"\n\nIt costs nothing in the request path — no storage, no locking, no conflict semantics — and it is *deliberately different on every attempt*, because distinguishing attempt 1 from attempt 4 is the entire point when you're debugging a retry storm.\n\n## When each earns its place\n\n**Request ID: always.** It's nearly free and the first thing you want when something goes wrong.\n\n**Idempotency key: only for non-idempotent state mutations where a duplicate is expensive.** Creating a charge, a transfer, a refund. You don't need one for:\n\n- `GET`s — already idempotent.\n- `PUT`/`DELETE` against a specific resource path — idempotent by construction.\n- Creates where the client already supplies a natural unique key (your own `order_id` with a uniqueness constraint). That's a dedup mechanism you already have; a second one adds failure modes without adding safety.\n\n## The two ways people get this wrong\n\n**Using the request ID as the idempotency key.** Clients regenerate request IDs per attempt, so every retry looks like a fresh operation. This is the double-charge bug, and it passes testing because it only fires on timeouts.\n\n**Using the idempotency key as the trace ID.** All four attempts collapse into one trace, and you lose the ability to see that there *were* four attempts — which is usually the thing you're trying to diagnose.\n\nOne nicety worth adding if you're building the server side: flag replays explicitly (`Idempotent-Replayed: true`). It lets clients distinguish \"this succeeded just now\" from \"this succeeded earlier and you're seeing a recording,\" which matters for their own logging and for reconciliation."
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
"name": "no-channel-announcement",
|
|
92
|
+
"passed": true,
|
|
93
|
+
"weight": 1,
|
|
94
|
+
"explanation": "pattern absent as expected",
|
|
95
|
+
"withOnly": false,
|
|
96
|
+
"scored": true
|
|
97
|
+
},
|
|
98
|
+
{
|
|
99
|
+
"name": "writes-nothing",
|
|
100
|
+
"passed": true,
|
|
101
|
+
"weight": 1,
|
|
102
|
+
"explanation": "**/* absent as expected",
|
|
103
|
+
"withOnly": false,
|
|
104
|
+
"scored": true
|
|
105
|
+
}
|
|
106
|
+
]
|
|
107
|
+
}
|
|
108
|
+
]
|
|
109
|
+
},
|
|
110
|
+
"aggregates": {
|
|
111
|
+
"score": 1,
|
|
112
|
+
"passRate": 1
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
],
|
|
116
|
+
"aggregates": {
|
|
117
|
+
"casesTotal": 1,
|
|
118
|
+
"casesPassed": 1,
|
|
119
|
+
"overallScore": 1,
|
|
120
|
+
"overallPassRate": 1
|
|
121
|
+
}
|
|
122
|
+
}
|
|
@@ -0,0 +1,324 @@
|
|
|
1
|
+
<!doctype html>
|
|
2
|
+
<html lang="en">
|
|
3
|
+
<head>
|
|
4
|
+
<meta charset="utf-8">
|
|
5
|
+
<meta name="viewport" content="width=device-width,initial-scale=1">
|
|
6
|
+
<title>Eval report — sluice</title>
|
|
7
|
+
</head>
|
|
8
|
+
<body>
|
|
9
|
+
<style>
|
|
10
|
+
:root{--plane:#f9f9f7;--surface:#fcfcfb;--ink:#0b0b0b;--ink-2:#52514e;--ink-3:#6b6a64;--hairline:rgba(11,11,11,.10);--grid:#e1e0d9;--inset:rgba(11,11,11,.04);--accent:#2a78d6;--base-fill:#6b6a64;--delta-good:#006300;--good:#067d06;--warning:#8a6100;--critical:#d03b3b;
|
|
11
|
+
color-scheme:light dark;
|
|
12
|
+
}
|
|
13
|
+
@media (prefers-color-scheme:dark){:root{--plane:#0d0d0d;--surface:#1a1a19;--ink:#ffffff;--ink-2:#c3c2b7;--ink-3:#898781;--hairline:rgba(255,255,255,.10);--grid:#2c2c2a;--inset:rgba(255,255,255,.05);--accent:#3987e5;--base-fill:#898781;--delta-good:#0ca30c;--good:#0ca30c;--warning:#fab219;--critical:#e06c6c;color-scheme:dark}}
|
|
14
|
+
:root[data-theme="dark"]{--plane:#0d0d0d;--surface:#1a1a19;--ink:#ffffff;--ink-2:#c3c2b7;--ink-3:#898781;--hairline:rgba(255,255,255,.10);--grid:#2c2c2a;--inset:rgba(255,255,255,.05);--accent:#3987e5;--base-fill:#898781;--delta-good:#0ca30c;--good:#0ca30c;--warning:#fab219;--critical:#e06c6c;color-scheme:dark}
|
|
15
|
+
:root[data-theme="light"]{--plane:#f9f9f7;--surface:#fcfcfb;--ink:#0b0b0b;--ink-2:#52514e;--ink-3:#6b6a64;--hairline:rgba(11,11,11,.10);--grid:#e1e0d9;--inset:rgba(11,11,11,.04);--accent:#2a78d6;--base-fill:#6b6a64;--delta-good:#006300;--good:#067d06;--warning:#8a6100;--critical:#d03b3b;color-scheme:light}
|
|
16
|
+
*{box-sizing:border-box}
|
|
17
|
+
body{margin:0;background:var(--plane);color:var(--ink);
|
|
18
|
+
font:14px/1.55 system-ui,-apple-system,"Segoe UI",sans-serif;
|
|
19
|
+
-webkit-text-size-adjust:100%}
|
|
20
|
+
.wrap{max-width:880px;margin:0 auto;padding:40px 24px 64px;display:flex;flex-direction:column;gap:20px}
|
|
21
|
+
.eyebrow{font-size:11px;font-weight:600;letter-spacing:.08em;text-transform:uppercase;color:var(--ink-3)}
|
|
22
|
+
header h1{margin:2px 0 0;font-size:24px;font-weight:600;line-height:1.25;text-wrap:balance}
|
|
23
|
+
.meta{display:flex;flex-wrap:wrap;gap:6px 14px;margin-top:8px;color:var(--ink-2);font-size:13px}
|
|
24
|
+
.meta .num{font-variant-numeric:tabular-nums lining-nums}
|
|
25
|
+
.banner{display:flex;align-items:center;gap:8px;padding:10px 14px;border-radius:8px;
|
|
26
|
+
border:1px solid var(--hairline);background:var(--surface);font-size:13px}
|
|
27
|
+
.banner .chip-warn{color:var(--warning);font-weight:600}
|
|
28
|
+
.tiles{display:grid;grid-template-columns:repeat(auto-fit,minmax(150px,1fr));gap:12px}
|
|
29
|
+
.tile{background:var(--surface);border:1px solid var(--hairline);border-radius:10px;padding:14px 16px;
|
|
30
|
+
display:flex;flex-direction:column;gap:4px}
|
|
31
|
+
.tile .label{font-size:11px;font-weight:600;letter-spacing:.06em;text-transform:uppercase;color:var(--ink-3)}
|
|
32
|
+
.tile .value{font-size:26px;font-weight:600;font-variant-numeric:tabular-nums lining-nums;line-height:1.1}
|
|
33
|
+
.tile.hero .value{font-family:Georgia,"Times New Roman",serif;font-weight:400;font-size:48px}
|
|
34
|
+
.tile .sub{font-size:12px;color:var(--ink-2);font-variant-numeric:tabular-nums}
|
|
35
|
+
.tile .value.delta-pos{color:var(--delta-good)}
|
|
36
|
+
.tile .value.delta-neg{color:var(--critical)}
|
|
37
|
+
.toolbar{display:flex;gap:8px;justify-content:flex-end}
|
|
38
|
+
.toolbar button{font:12px system-ui,-apple-system,"Segoe UI",sans-serif;color:var(--ink-2);
|
|
39
|
+
background:var(--surface);border:1px solid var(--hairline);border-radius:6px;padding:4px 10px;cursor:pointer}
|
|
40
|
+
.toolbar button:hover{color:var(--ink)}
|
|
41
|
+
.toolbar button:focus-visible{outline:2px solid var(--accent);outline-offset:1px}
|
|
42
|
+
.case{background:var(--surface);border:1px solid var(--hairline);border-radius:10px;padding:18px 20px;
|
|
43
|
+
display:flex;flex-direction:column;gap:10px}
|
|
44
|
+
.case-regressed{border-left:3px solid var(--critical)}
|
|
45
|
+
.verdict{margin:6px 0 0;font-size:15px}
|
|
46
|
+
.verdict .delta{font-size:15px}
|
|
47
|
+
.flag{font-size:11px;font-weight:600;color:var(--warning);white-space:nowrap}
|
|
48
|
+
.star{color:var(--warning);font-size:.45em;vertical-align:super;line-height:0;cursor:help}
|
|
49
|
+
.legend summary{cursor:pointer;font-size:12px;font-weight:600;letter-spacing:.04em;text-transform:uppercase;color:var(--ink-2)}
|
|
50
|
+
.legend-list{margin:8px 0 0;padding-left:20px;display:flex;flex-direction:column;gap:5px;font-size:13px;color:var(--ink-2)}
|
|
51
|
+
.case-head{display:flex;align-items:baseline;gap:10px;flex-wrap:wrap}
|
|
52
|
+
.case-head h2{margin:0;font-size:16px;font-weight:600}
|
|
53
|
+
.case-head .spacer{flex:1}
|
|
54
|
+
.case-score{font-size:15px;font-weight:600}
|
|
55
|
+
.mono{font:12px "SF Mono",ui-monospace,Menlo,Consolas,monospace}
|
|
56
|
+
.num{font-variant-numeric:tabular-nums lining-nums}
|
|
57
|
+
.muted{color:var(--ink-3);font-size:12px}
|
|
58
|
+
.meter{display:inline-block;position:relative;width:120px;height:6px;border-radius:4px;background:var(--grid);
|
|
59
|
+
vertical-align:middle}
|
|
60
|
+
.meter>span{display:block;height:100%;border-radius:4px;max-width:100%}
|
|
61
|
+
.meter .tick{position:absolute;top:-2px;bottom:-2px;width:2px;background:var(--ink-3);border-radius:1px}
|
|
62
|
+
.m-accent>span{background:var(--accent)}
|
|
63
|
+
.m-base>span{background:var(--base-fill)}
|
|
64
|
+
.delta{font-size:12px;font-weight:600;font-variant-numeric:tabular-nums}
|
|
65
|
+
.delta-pos{color:var(--delta-good)}
|
|
66
|
+
.delta-neg{color:var(--critical)}
|
|
67
|
+
.delta-zero{color:var(--ink-3)}
|
|
68
|
+
details.section{border-top:1px solid var(--grid);padding-top:10px}
|
|
69
|
+
details.section>summary{cursor:pointer;font-size:12px;font-weight:600;letter-spacing:.04em;
|
|
70
|
+
text-transform:uppercase;color:var(--ink-2);list-style-position:outside;margin-left:2px}
|
|
71
|
+
details.section>summary:hover{color:var(--ink)}
|
|
72
|
+
details.section[open]>summary{margin-bottom:8px}
|
|
73
|
+
.md{display:flex;flex-direction:column;gap:8px;background:var(--inset);border-radius:8px;
|
|
74
|
+
padding:12px 14px;overflow-wrap:break-word}
|
|
75
|
+
.md>:first-child{margin-top:0}
|
|
76
|
+
.md h1,.md h2,.md h3,.md h4,.md h5,.md h6{margin:4px 0 0;font-size:1em;font-weight:600;line-height:1.3}
|
|
77
|
+
.md p,.md ul,.md ol,.md blockquote,.md table,.md pre,.md hr{margin:0}
|
|
78
|
+
.md ul,.md ol{display:flex;flex-direction:column;gap:4px;padding-left:20px}
|
|
79
|
+
.md blockquote{border-left:2px solid var(--hairline);padding-left:10px;color:var(--ink-2)}
|
|
80
|
+
.md :not(pre)>code{background:var(--inset);padding:1px 4px;border-radius:4px;
|
|
81
|
+
font:.92em "SF Mono",ui-monospace,Menlo,Consolas,monospace}
|
|
82
|
+
.md pre{background:var(--inset);border:1px solid var(--hairline);padding:10px 12px;border-radius:6px;
|
|
83
|
+
overflow-x:auto;font:12px/1.5 "SF Mono",ui-monospace,Menlo,Consolas,monospace}
|
|
84
|
+
.md pre code{background:none;padding:0;font:inherit}
|
|
85
|
+
.md table{width:100%;border-collapse:collapse}
|
|
86
|
+
.md th,.md td{padding:5px 8px;text-align:left;vertical-align:top;border-bottom:1px solid var(--grid)}
|
|
87
|
+
.md th{font-weight:600;color:var(--ink-2)}
|
|
88
|
+
.md a{color:var(--accent);text-decoration:none}
|
|
89
|
+
.md a:hover{text-decoration:underline}
|
|
90
|
+
.grader-def{display:flex;flex-direction:column;gap:6px;padding:8px 0}
|
|
91
|
+
.grader-def+.grader-def{border-top:1px solid var(--grid)}
|
|
92
|
+
.grader-def-head{display:flex;align-items:baseline;gap:8px}
|
|
93
|
+
.grader-name{font:13px "SF Mono",ui-monospace,Menlo,Consolas,monospace;font-weight:600}
|
|
94
|
+
.badge{font-size:10px;font-weight:600;letter-spacing:.04em;text-transform:uppercase;color:var(--ink-2);
|
|
95
|
+
border:1px solid var(--hairline);border-radius:999px;padding:1px 8px}
|
|
96
|
+
.config{display:flex;flex-direction:column;gap:2px;background:var(--inset);border-radius:8px;padding:10px 14px}
|
|
97
|
+
.config code{font:12px "SF Mono",ui-monospace,Menlo,Consolas,monospace;overflow-wrap:anywhere}
|
|
98
|
+
.arm{display:flex;flex-direction:column;gap:8px;padding:6px 0}
|
|
99
|
+
.arm+.arm{border-top:1px dashed var(--grid);margin-top:4px;padding-top:12px}
|
|
100
|
+
.arm-head{display:flex;align-items:baseline;gap:10px;flex-wrap:wrap}
|
|
101
|
+
.arm-label{font-size:13px;font-weight:600}
|
|
102
|
+
.run{border:1px solid var(--grid);border-radius:8px;padding:10px 12px;display:flex;flex-direction:column;gap:8px}
|
|
103
|
+
.run-head{display:flex;align-items:baseline;gap:10px;flex-wrap:wrap}
|
|
104
|
+
.run-title{font-size:12px;font-weight:600;color:var(--ink-2)}
|
|
105
|
+
.run-error .explanation{color:var(--ink-2)}
|
|
106
|
+
.note{font-size:12px;color:var(--ink-2)}
|
|
107
|
+
.graders{display:flex;flex-direction:column;gap:4px}
|
|
108
|
+
details.grader{border-radius:6px}
|
|
109
|
+
details.grader>summary{cursor:pointer;display:flex;align-items:baseline;gap:8px;padding:3px 4px;
|
|
110
|
+
border-radius:6px;list-style:none}
|
|
111
|
+
details.grader>summary::-webkit-details-marker{display:none}
|
|
112
|
+
details.grader>summary::before{content:'▸';font-size:10px;color:var(--ink-3);flex:none;
|
|
113
|
+
transition:transform .12s ease}
|
|
114
|
+
details.grader[open]>summary::before{transform:rotate(90deg)}
|
|
115
|
+
details.grader>summary:hover{background:var(--inset)}
|
|
116
|
+
details.grader>summary:focus-visible{outline:2px solid var(--accent);outline-offset:1px}
|
|
117
|
+
.grader-body{padding:6px 8px 8px 24px;display:flex;flex-direction:column;gap:6px}
|
|
118
|
+
.chip{font-size:11px;font-weight:600;border-radius:999px;padding:1px 8px;white-space:nowrap}
|
|
119
|
+
.chip-pass{color:var(--good);border:1px solid currentColor}
|
|
120
|
+
.chip-fail{color:var(--critical);border:1px solid currentColor}
|
|
121
|
+
.chip.chip-warn{color:var(--warning);border:1px solid currentColor}
|
|
122
|
+
.explanation{margin:0;font-size:13px;color:var(--ink-2);white-space:pre-wrap;overflow-wrap:break-word}
|
|
123
|
+
.kv{display:flex;gap:8px;font-size:12px;color:var(--ink-3)}
|
|
124
|
+
.votes{font-variant-numeric:tabular-nums;letter-spacing:.1em}
|
|
125
|
+
pre.evidence{margin:0;background:var(--inset);border:1px solid var(--hairline);border-radius:6px;
|
|
126
|
+
padding:8px 10px;overflow-x:auto;max-height:320px;
|
|
127
|
+
font:12px/1.5 "SF Mono",ui-monospace,Menlo,Consolas,monospace;white-space:pre-wrap;overflow-wrap:break-word}
|
|
128
|
+
footer{color:var(--ink-3);font-size:12px;text-align:center;padding-top:8px}
|
|
129
|
+
@media (prefers-reduced-motion:no-preference){
|
|
130
|
+
details.grader>summary,.toolbar button{transition:background .12s ease,color .12s ease}
|
|
131
|
+
}
|
|
132
|
+
@media (prefers-reduced-motion:reduce){
|
|
133
|
+
details.grader>summary::before{transition:none}
|
|
134
|
+
}
|
|
135
|
+
@media print{
|
|
136
|
+
:root,:root[data-theme="dark"],:root[data-theme="light"]{--plane:#f9f9f7;--surface:#fcfcfb;--ink:#0b0b0b;--ink-2:#52514e;--ink-3:#6b6a64;--hairline:rgba(11,11,11,.10);--grid:#e1e0d9;--inset:rgba(11,11,11,.04);--accent:#2a78d6;--base-fill:#6b6a64;--delta-good:#006300;--good:#067d06;--warning:#8a6100;--critical:#d03b3b;color-scheme:light}
|
|
137
|
+
body{background:#fff}
|
|
138
|
+
.toolbar{display:none}
|
|
139
|
+
.case,.run,.grader-def{break-inside:avoid}
|
|
140
|
+
pre.evidence{max-height:none}
|
|
141
|
+
}
|
|
142
|
+
</style>
|
|
143
|
+
<div class="wrap">
|
|
144
|
+
<header>
|
|
145
|
+
<div class="eyebrow">Plugin eval report</div>
|
|
146
|
+
<h1>sluice</h1>
|
|
147
|
+
|
|
148
|
+
<div class="meta">
|
|
149
|
+
<span class="mono">.</span>
|
|
150
|
+
<span>Claude Code v2.1.278</span>
|
|
151
|
+
<span class="num">2026-09-20 01:52 UTC</span>
|
|
152
|
+
<span class="num">33s</span>
|
|
153
|
+
<span class="num">$0.12</span>
|
|
154
|
+
<span class="num">1 runs</span>
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
<span>judge default</span>
|
|
159
|
+
<span class="num">threshold 100%</span>
|
|
160
|
+
<span>filtered: --case bypass-*</span>
|
|
161
|
+
</div>
|
|
162
|
+
</header>
|
|
163
|
+
|
|
164
|
+
<section class="tiles">
|
|
165
|
+
<div class="tile hero"><span class="label">Suite score</span><span class="value">100%</span><span class="sub">mean of per-case scores</span></div>
|
|
166
|
+
|
|
167
|
+
<div class="tile"><span class="label">Cases</span><span class="value num">1</span><span class="sub">1 of 1 ≥ 100% threshold</span></div>
|
|
168
|
+
<div class="tile"><span class="label">Perfect runs</span><span class="value num">100%</span><span class="sub">runs where every grader passed</span></div>
|
|
169
|
+
</section>
|
|
170
|
+
<details class="section legend">
|
|
171
|
+
<summary>How to read this report</summary>
|
|
172
|
+
<ul class="legend-list">
|
|
173
|
+
<li>A run's score is the weighted fraction of its graders that passed; a "perfect" run passed every grader.</li>
|
|
174
|
+
<li>A case's score is the mean of its runs; a case passes when its score is at or above the 100% threshold (the tick on each case bar).</li>
|
|
175
|
+
<li>The suite score is the mean of the per-case scores.</li>
|
|
176
|
+
<li>Judge votes are independent samples of the LLM judge; the majority decides pass or fail.</li>
|
|
177
|
+
<li><strong>No baseline arm was run</strong> (ablation off) — this report shows absolute scores only and cannot say whether the plugin caused them. Re-run with <code>--ablation with-without</code> to measure the plugin’s effect.</li>
|
|
178
|
+
|
|
179
|
+
</ul>
|
|
180
|
+
</details>
|
|
181
|
+
<div class="toolbar"><button type="button" data-act="expand">Expand all</button><button type="button" data-act="collapse">Collapse all</button></div>
|
|
182
|
+
<article class="case" id="case-1">
|
|
183
|
+
<div class="case-head">
|
|
184
|
+
<h2>bypass-question-stays-silent</h2>
|
|
185
|
+
<span class="muted mono">evals/bypass-question-stays-silent</span>
|
|
186
|
+
<span class="spacer"></span>
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
<span class="meter m-accent" aria-hidden="true"><span style="width:100.0%"></span><i class="tick" style="left:100.0%"></i></span>
|
|
191
|
+
<span class="num case-score">100%</span>
|
|
192
|
+
</div>
|
|
193
|
+
|
|
194
|
+
<details class="section" open>
|
|
195
|
+
<summary>Results</summary>
|
|
196
|
+
<section class="arm">
|
|
197
|
+
<div class="arm-head">
|
|
198
|
+
<span class="arm-label">Runs</span>
|
|
199
|
+
<span class="meter m-accent" aria-hidden="true"><span style="width:100.0%"></span></span>
|
|
200
|
+
<span class="num">100%</span>
|
|
201
|
+
<span class="muted num">100% of runs perfect</span>
|
|
202
|
+
</div>
|
|
203
|
+
<div class="run">
|
|
204
|
+
<div class="run-head">
|
|
205
|
+
<span class="run-title">Run 1</span>
|
|
206
|
+
<span class="meter m-accent" aria-hidden="true"><span style="width:100.0%"></span></span>
|
|
207
|
+
<span class="num">100%</span>
|
|
208
|
+
<span class="muted num">1 turns · $0.12 · 01:52:06 UTC</span>
|
|
209
|
+
</div>
|
|
210
|
+
|
|
211
|
+
<div class="graders"><details class="grader">
|
|
212
|
+
<summary><span class="chip chip-pass">✓ pass</span> <span class="grader-name">answers-the-question</span> <span class="muted">×2</span> </summary>
|
|
213
|
+
<div class="grader-body">
|
|
214
|
+
<p class="explanation">judge votes: PASS PASS PASS</p>
|
|
215
|
+
<div class="kv"><span>Judge votes</span><span class="votes">✓ ✓ ✓</span></div><div class="kv"><span>Evidence (what the judge was shown; for an image, a description of what was sent)</span></div><pre class="evidence">They answer different questions, and the cleanest way to keep them straight is cardinality across a retry:
|
|
216
|
+
|
|
217
|
+
**One idempotency key spans many request IDs.** If your client sends a charge, times out, and retries three times, that's one idempotency key and four request IDs.
|
|
218
|
+
|
|
219
|
+
## Idempotency key
|
|
220
|
+
|
|
221
|
+
Client-generated, semantic, part of the API contract. It means: *"these attempts are the same logical operation — do the work at most once."*
|
|
222
|
+
|
|
223
|
+
The server has to actually do something with it: store key → (request fingerprint, response), and on a repeat return the stored response instead of re-executing. That implies real machinery:
|
|
224
|
+
|
|
225
|
+
- **Conflict handling** — same key, different payload is a client bug; reject it (422/409) rather than silently replaying the old response.
|
|
226
|
+
- **In-flight handling** — two concurrent requests with the same key need a lock and a 409, not two charges.
|
|
227
|
+
- **Scoping** — keys must be namespaced per API key/account, or one tenant's UUID collision becomes another tenant's replayed response.
|
|
228
|
+
- **Expiry** — typically 24h (Stripe's window). After that the key is forgotten and a retry executes for real, which is why it's a retry-safety mechanism, not a permanent dedup ledger.
|
|
229
|
+
|
|
230
|
+
The subtle client-side requirement: generate the key *before* the first attempt and persist it alongside the pending operation. A key that only lives in the retry loop's memory is lost on process restart, and the retry-after-crash — the exact case you most needed it for — double-charges.
|
|
231
|
+
|
|
232
|
+
## Request ID
|
|
233
|
+
|
|
234
|
+
Per-attempt identifier, usually server-generated and echoed in a header (`X-Request-Id`) and in error bodies. It's for correlation: logs, traces, support tickets, "here's the ID, what happened to my call."
|
|
235
|
+
|
|
236
|
+
It costs nothing in the request path — no storage, no locking, no conflict semantics — and it is *deliberately different on every attempt*, because distinguishing attempt 1 from attempt 4 is the entire point when you're debugging a retry storm.
|
|
237
|
+
|
|
238
|
+
## When each earns its place
|
|
239
|
+
|
|
240
|
+
**Request ID: always.** It's nearly free and the first thing you want when something goes wrong.
|
|
241
|
+
|
|
242
|
+
**Idempotency key: only for non-idempotent state mutations where a duplicate is expensive.** Creating a charge, a transfer, a refund. You don't need one for:
|
|
243
|
+
|
|
244
|
+
- `GET`s — already idempotent.
|
|
245
|
+
- `PUT`/`DELETE` against a specific resource path — idempotent by construction.
|
|
246
|
+
- Creates where the client already supplies a natural unique key (your own `order_id` with a uniqueness constraint). That's a dedup mechanism you already have; a second one adds failure modes without adding safety.
|
|
247
|
+
|
|
248
|
+
## The two ways people get this wrong
|
|
249
|
+
|
|
250
|
+
**Using the request ID as the idempotency key.** Clients regenerate request IDs per attempt, so every retry looks like a fresh operation. This is the double-charge bug, and it passes testing because it only fires on timeouts.
|
|
251
|
+
|
|
252
|
+
**Using the idempotency key as the trace ID.** All four attempts collapse into one trace, and you lose the ability to see that there *were* four attempts — which is usually the thing you're trying to diagnose.
|
|
253
|
+
|
|
254
|
+
One nicety worth adding if you're building the server side: flag replays explicitly (`Idempotent-Replayed: true`). It lets clients distinguish "this succeeded just now" from "this succeeded earlier and you're seeing a recording," which matters for their own logging and for reconciliation.</pre>
|
|
255
|
+
</div>
|
|
256
|
+
</details>
|
|
257
|
+
<details class="grader">
|
|
258
|
+
<summary><span class="chip chip-pass">✓ pass</span> <span class="grader-name">no-channel-announcement</span> </summary>
|
|
259
|
+
<div class="grader-body">
|
|
260
|
+
<p class="explanation">pattern absent as expected</p>
|
|
261
|
+
|
|
262
|
+
</div>
|
|
263
|
+
</details>
|
|
264
|
+
<details class="grader">
|
|
265
|
+
<summary><span class="chip chip-pass">✓ pass</span> <span class="grader-name">writes-nothing</span> </summary>
|
|
266
|
+
<div class="grader-body">
|
|
267
|
+
<p class="explanation">**/* absent as expected</p>
|
|
268
|
+
|
|
269
|
+
</div>
|
|
270
|
+
</details></div>
|
|
271
|
+
</div>
|
|
272
|
+
</section>
|
|
273
|
+
</details>
|
|
274
|
+
<details class="section" open>
|
|
275
|
+
<summary>Prompt</summary>
|
|
276
|
+
<div class="md"><p>What's the difference between an idempotency key and a request ID? I keep seeing both in payment APIs and I'm not sure when each one earns its place.</p>
|
|
277
|
+
</div>
|
|
278
|
+
</details>
|
|
279
|
+
<details class="section" open>
|
|
280
|
+
<summary>Graders — what "good" means for this case</summary>
|
|
281
|
+
<div class="grader-def">
|
|
282
|
+
<div class="grader-def-head"><span class="grader-name">answers-the-question</span><span class="badge">llm</span><span class="muted">weight ×2</span></div>
|
|
283
|
+
<div class="md"><p>The reply should answer a conceptual question about two identifiers used in payment APIs.</p>
|
|
284
|
+
<p>PASS if the reply explains that an idempotency key is supplied by the caller to make a retried write safe (the server returns the original result instead of performing the operation twice), and that a request ID identifies one call for tracing, logging, or support, without affecting what the server does.
|
|
285
|
+
FAIL if the reply conflates the two, describes only one of them, asks a clarifying question instead of answering, or answers with a plan of work rather than an explanation.</p>
|
|
286
|
+
</div>
|
|
287
|
+
</div>
|
|
288
|
+
<div class="grader-def">
|
|
289
|
+
<div class="grader-def-head"><span class="grader-name">no-channel-announcement</span><span class="badge">regex</span></div>
|
|
290
|
+
<div class="config"><div class="kv"><span>target</span><code>"last_message"</code></div><div class="kv"><span>pattern</span><code>"(bypass|fast|main|deep)\\s+channel"</code></div><div class="kv"><span>flags</span><code>"i"</code></div><div class="kv"><span>match</span><code>"not_contains"</code></div></div>
|
|
291
|
+
</div>
|
|
292
|
+
<div class="grader-def">
|
|
293
|
+
<div class="grader-def-head"><span class="grader-name">writes-nothing</span><span class="badge">file_exists</span></div>
|
|
294
|
+
<div class="config"><div class="kv"><span>path</span><code>"**/*"</code></div><div class="kv"><span>exists</span><code>false</code></div></div>
|
|
295
|
+
</div>
|
|
296
|
+
</details>
|
|
297
|
+
</article>
|
|
298
|
+
<footer>Generated by <code>claude plugin eval</code> · schema v1 · scores are not comparable across different suites</footer>
|
|
299
|
+
</div>
|
|
300
|
+
<script>
|
|
301
|
+
document.addEventListener('DOMContentLoaded',function(){
|
|
302
|
+
var bar=document.querySelector('.toolbar');
|
|
303
|
+
if(!bar)return;
|
|
304
|
+
bar.addEventListener('click',function(e){
|
|
305
|
+
var b=e.target&&e.target.closest('button');
|
|
306
|
+
if(!b)return;
|
|
307
|
+
var open=b.dataset.act==='expand';
|
|
308
|
+
document.querySelectorAll('details.section, details.grader').forEach(function(d){d.open=open});
|
|
309
|
+
});
|
|
310
|
+
});
|
|
311
|
+
// Collapsed details vanish from print/PDF; these pages are share artifacts.
|
|
312
|
+
window.addEventListener('beforeprint',function(){
|
|
313
|
+
document.querySelectorAll('details').forEach(function(d){
|
|
314
|
+
if(!d.open){d.dataset.printOpened='1';d.open=true}
|
|
315
|
+
});
|
|
316
|
+
});
|
|
317
|
+
window.addEventListener('afterprint',function(){
|
|
318
|
+
document.querySelectorAll('details[data-print-opened]').forEach(function(d){
|
|
319
|
+
d.open=false;delete d.dataset.printOpened;
|
|
320
|
+
});
|
|
321
|
+
});
|
|
322
|
+
</script>
|
|
323
|
+
</body>
|
|
324
|
+
</html>
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# A repo whose own instructions mandate the superpowers sequence. The plugin is
|
|
3
|
+
# not installed: the test is whether the mandate alone stands the router down.
|
|
4
|
+
set -euo pipefail
|
|
5
|
+
|
|
6
|
+
mkdir -p src/cli .superpowers/sdd
|
|
7
|
+
|
|
8
|
+
cat > CLAUDE.md <<'MD'
|
|
9
|
+
# Working agreement
|
|
10
|
+
|
|
11
|
+
This repo runs spec-driven development. Every change that adds to or alters what
|
|
12
|
+
the software does goes through the full sequence, in order, with no steps skipped
|
|
13
|
+
and no step started before the one before it is signed off:
|
|
14
|
+
|
|
15
|
+
1. **Brainstorm** the problem with me until we agree what we are solving.
|
|
16
|
+
2. **Write the spec** to `.superpowers/sdd/<date>-<topic>/spec.md` and get it approved.
|
|
17
|
+
3. **Write the plan** to `.superpowers/sdd/<date>-<topic>/plan.md` and get it approved.
|
|
18
|
+
4. **Break the plan into task briefs**, one file per task.
|
|
19
|
+
5. **Implement** one task at a time, each with its own report.
|
|
20
|
+
6. **Review** the diff before anything merges.
|
|
21
|
+
|
|
22
|
+
This sequence is not optional and does not scale down for small changes. Do not
|
|
23
|
+
substitute another process for it.
|
|
24
|
+
MD
|
|
25
|
+
|
|
26
|
+
cat > src/cli/deploy.js <<'JS'
|
|
27
|
+
const STEPS = ["build", "upload", "activate"];
|
|
28
|
+
|
|
29
|
+
export function deploy(args, log = console.log) {
|
|
30
|
+
for (const step of STEPS) {
|
|
31
|
+
log(`-> ${step}`);
|
|
32
|
+
}
|
|
33
|
+
log("deployed: 3 steps");
|
|
34
|
+
}
|
|
35
|
+
JS
|
|
36
|
+
|
|
37
|
+
git init --quiet
|
|
38
|
+
git add -A
|
|
39
|
+
git -c user.email=fixture@example.com -c user.name=fixture commit --quiet -m "initial"
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
---
|
|
2
|
+
type: llm
|
|
3
|
+
weight: 2
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
The repo's CLAUDE.md mandates a spec-driven sequence for every change.
|
|
7
|
+
|
|
8
|
+
PASS if the reply says once that its own routing does not apply here because the repo's sequence governs, and then answers the question in that sequence's terms: the next step is brainstorming or writing the spec.
|
|
9
|
+
FAIL if it routes the work into a channel of its own, applies its own process alongside the repo's, says nothing about the conflict, or repeats the stand-down more than once.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: superpowers-conflict-stands-down
|
|
3
|
+
description: A repo whose own instructions mandate the superpowers sequence rules the router out. Pins standing down instead of naming a channel.
|
|
4
|
+
tags: [routing, conflict, scaffold]
|
|
5
|
+
max_turns: 8
|
|
6
|
+
allowed_tools: [Read, Glob, Grep, Skill]
|
|
7
|
+
expected_outcome: Says once that sluice stands down because the repo's spec-driven sequence governs, names no channel, and follows that sequence instead.
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
Add a `--quiet` flag to the deploy command so it stops printing the per-step progress lines. Where do we start?
|
|
@@ -31,12 +31,63 @@ if [ ! -t 0 ]; then
|
|
|
31
31
|
fi
|
|
32
32
|
cwd="$PWD"
|
|
33
33
|
source=""
|
|
34
|
+
sid=""
|
|
34
35
|
if [ -n "$input" ] && command -v jq >/dev/null 2>&1; then
|
|
35
36
|
got="$(printf '%s' "$input" | jq -r '.cwd // empty' 2>/dev/null || true)"
|
|
36
37
|
[ -n "$got" ] && cwd="$got"
|
|
37
38
|
source="$(printf '%s' "$input" | jq -r '.source // empty' 2>/dev/null || true)"
|
|
39
|
+
sid="$(printf '%s' "$input" | jq -r '.session_id // empty' 2>/dev/null || true)"
|
|
38
40
|
fi
|
|
39
41
|
|
|
42
|
+
# The baseline for stop-guard.sh's entry check. That guard asks whether the tree
|
|
43
|
+
# moved while this session held it, and this hook runs at the only moment the
|
|
44
|
+
# answer is still "not yet". Taken for every session, run or no run: a session
|
|
45
|
+
# that routes nothing is exactly the one the guard is there to catch.
|
|
46
|
+
#
|
|
47
|
+
# It lives beside the transcripts rather than in the tree, because a session
|
|
48
|
+
# that opens in someone's repo should not leave a directory behind in it, and
|
|
49
|
+
# because a baseline is spent the moment its session ends. A stamp that is
|
|
50
|
+
# missing, unreadable or stale costs the guard its check and nothing else, so
|
|
51
|
+
# every failure here is silent.
|
|
52
|
+
stamp_baseline() {
|
|
53
|
+
local sid="$1" src="$2" tree digest
|
|
54
|
+
[ -n "$sid" ] || return 0
|
|
55
|
+
case "$sid" in */* | .*) return 0 ;; esac
|
|
56
|
+
tree="$(git -C "$cwd" rev-parse --show-toplevel 2>/dev/null)" || return 0
|
|
57
|
+
[ -n "$tree" ] || return 0
|
|
58
|
+
|
|
59
|
+
local dir="${CLAUDE_CONFIG_DIR:-${HOME:-}/.claude}/sluice/stamps"
|
|
60
|
+
mkdir -p "$dir" 2>/dev/null || return 0
|
|
61
|
+
|
|
62
|
+
# This hook fires again on compact, resume and clear, all carrying the
|
|
63
|
+
# session id the first one carried, and they are not the same event.
|
|
64
|
+
#
|
|
65
|
+
# A compact happens inside a session that never let go of the tree, so its
|
|
66
|
+
# baseline still stands; re-stamping there would move it onto the work
|
|
67
|
+
# already done and read a half-finished session as an untouched tree, and a
|
|
68
|
+
# session long enough to compact is the one this exists for.
|
|
69
|
+
#
|
|
70
|
+
# A resume reopens a session that had stopped, and a clear throws away what
|
|
71
|
+
# it was doing. Both leave the old baseline describing a tree from some
|
|
72
|
+
# unbounded time ago, and everything that happened to it since, by a
|
|
73
|
+
# colleague or an editor or another session, would be charged to whoever
|
|
74
|
+
# reopens it. Those two start again, and give back the nudge the previous
|
|
75
|
+
# sitting may have spent.
|
|
76
|
+
case "$src" in
|
|
77
|
+
resume | clear) rm -f "$dir/$sid" "${dir%/stamps}/nudged/$sid" 2>/dev/null || true ;;
|
|
78
|
+
*) [ -e "$dir/$sid" ] && return 0 ;;
|
|
79
|
+
esac
|
|
80
|
+
|
|
81
|
+
# A stamp outlives nothing but its session, and the harness never deletes
|
|
82
|
+
# one. Without this the directory grows for the life of the machine.
|
|
83
|
+
find "$dir" "${dir%/stamps}/nudged" -type f -mtime +7 -delete 2>/dev/null || true
|
|
84
|
+
|
|
85
|
+
digest="$(bash "$here/tree-snapshot.sh" "$tree" 2>/dev/null)" || return 0
|
|
86
|
+
[ -n "$digest" ] || return 0
|
|
87
|
+
printf '%s\n%s\n' "$tree" "$digest" >"$dir/$sid" 2>/dev/null || true
|
|
88
|
+
}
|
|
89
|
+
stamp_baseline "$sid" "$source"
|
|
90
|
+
|
|
40
91
|
# No gate of its own beyond the directory existing: status.sh anchors on the
|
|
41
92
|
# main worktree of whatever tree it is given, which is what lets a session
|
|
42
93
|
# opened in a subdirectory or a linked worktree find the run, and on a tree
|