luciazero 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +423 -0
- package/LICENSE +21 -0
- package/README.md +219 -0
- package/README.th.md +215 -0
- package/bin/luciazero.js +35 -0
- package/claude/agents/reviewer.md +43 -0
- package/claude/hooks/hooks.json +33 -0
- package/claude/hooks/luciazero-statusline.sh +81 -0
- package/claude/hooks/luciazero-verify.sh +252 -0
- package/claude/luciazero.md +24 -0
- package/install-codex.sh +84 -0
- package/install.sh +249 -0
- package/package.json +34 -0
- package/skills/debug/SKILL.md +51 -0
- package/skills/done/SKILL.md +57 -0
- package/skills/done/scripts/revert-probe.sh +95 -0
- package/skills/experiment/SKILL.md +44 -0
- package/skills/handoff/SKILL.md +56 -0
- package/skills/luciazero-bootstrap/SKILL.md +109 -0
- package/skills/luciazero-bootstrap/scripts/detect.sh +84 -0
- package/skills/retro/SKILL.md +74 -0
- package/uninstall-codex.sh +50 -0
- package/uninstall.sh +139 -0
package/install.sh
ADDED
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Install the Luciazero doctrine + skills into ~/.claude/
|
|
3
|
+
# Idempotent. Backs up CLAUDE.md (and settings.json when --with-hooks)
|
|
4
|
+
# before editing. Writes nothing outside ~/.claude/.
|
|
5
|
+
#
|
|
6
|
+
# ./install.sh doctrine + skills + reviewer agent
|
|
7
|
+
# ./install.sh --with-hooks also wire the enforcement pack: verify-tracking
|
|
8
|
+
# hooks + statusline into ~/.claude/settings.json
|
|
9
|
+
# (Claude Code only; requires python3)
|
|
10
|
+
# ./install.sh --status read-only health check of an existing install;
|
|
11
|
+
# exits non-zero if a core piece is missing
|
|
12
|
+
set -euo pipefail
|
|
13
|
+
|
|
14
|
+
WITH_HOOKS=0
|
|
15
|
+
STATUS_ONLY=0
|
|
16
|
+
for ARG in "$@"; do
|
|
17
|
+
case "${ARG}" in
|
|
18
|
+
--with-hooks) WITH_HOOKS=1 ;;
|
|
19
|
+
--status) STATUS_ONLY=1 ;;
|
|
20
|
+
*) echo "unknown option: ${ARG} (supported: --with-hooks, --status)" >&2; exit 1 ;;
|
|
21
|
+
esac
|
|
22
|
+
done
|
|
23
|
+
|
|
24
|
+
SRC="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
25
|
+
CLAUDE_DIR="${CLAUDE_CONFIG_DIR:-$HOME/.claude}"
|
|
26
|
+
DOCTRINE="luciazero.md"
|
|
27
|
+
IMPORT_LINE="@${DOCTRINE}"
|
|
28
|
+
|
|
29
|
+
# newest released version in this checkout's CHANGELOG (informational)
|
|
30
|
+
version_of() {
|
|
31
|
+
grep -m1 -oE '^## \[[0-9]+\.[0-9]+\.[0-9]+\]' "${SRC}/CHANGELOG.md" 2>/dev/null | tr -d '#[] ' || true
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
if [ "${STATUS_ONLY}" = 1 ]; then
|
|
35
|
+
echo "Status of ${CLAUDE_DIR} (read-only)"
|
|
36
|
+
STATUS_RC=0
|
|
37
|
+
check() { # check <file-test-flag: -f|-x> <path> <label...>
|
|
38
|
+
T="$1"; P="$2"; shift 2
|
|
39
|
+
OK=0
|
|
40
|
+
case "${T}" in
|
|
41
|
+
-x) if [ -x "${P}" ]; then OK=1; fi ;;
|
|
42
|
+
*) if [ -f "${P}" ]; then OK=1; fi ;;
|
|
43
|
+
esac
|
|
44
|
+
if [ "${OK}" = 1 ]; then echo " ok $*"; else echo " MISS $*"; STATUS_RC=1; fi
|
|
45
|
+
}
|
|
46
|
+
check -f "${CLAUDE_DIR}/${DOCTRINE}" "doctrine ${DOCTRINE}"
|
|
47
|
+
for SKILL in luciazero-bootstrap retro debug 'done' handoff experiment; do
|
|
48
|
+
check -f "${CLAUDE_DIR}/skills/${SKILL}/SKILL.md" "skill ${SKILL}"
|
|
49
|
+
done
|
|
50
|
+
check -x "${CLAUDE_DIR}/skills/luciazero-bootstrap/scripts/detect.sh" "detect.sh executable"
|
|
51
|
+
check -x "${CLAUDE_DIR}/skills/done/scripts/revert-probe.sh" "revert-probe.sh executable"
|
|
52
|
+
check -f "${CLAUDE_DIR}/agents/reviewer.md" "reviewer agent"
|
|
53
|
+
GLOBAL_MD="${CLAUDE_DIR}/CLAUDE.md"
|
|
54
|
+
N="$(grep -cxF "${IMPORT_LINE}" "${GLOBAL_MD}" 2>/dev/null || true)"
|
|
55
|
+
if [ "${N:-0}" = 1 ]; then
|
|
56
|
+
echo " ok CLAUDE.md imports the doctrine"
|
|
57
|
+
else
|
|
58
|
+
echo " MISS CLAUDE.md import line (${IMPORT_LINE} exactly once; found ${N:-0})"; STATUS_RC=1
|
|
59
|
+
fi
|
|
60
|
+
V_SRC="$(version_of)"
|
|
61
|
+
V_INST="$(cat "${CLAUDE_DIR}/.luciazero-version" 2>/dev/null || true)"
|
|
62
|
+
if [ -z "${V_INST}" ]; then
|
|
63
|
+
echo " -- installed version unknown (no sidecar — installed by an older version)"
|
|
64
|
+
elif [ "${V_INST}" = "${V_SRC}" ]; then
|
|
65
|
+
echo " ok version ${V_INST} (matches this checkout)"
|
|
66
|
+
else
|
|
67
|
+
echo " !! installed ${V_INST}, checkout ${V_SRC:-?} — re-run ./install.sh to update"
|
|
68
|
+
fi
|
|
69
|
+
if [ -f "${CLAUDE_DIR}/hooks/luciazero-verify.sh" ]; then
|
|
70
|
+
check -x "${CLAUDE_DIR}/hooks/luciazero-verify.sh" "hook luciazero-verify.sh executable"
|
|
71
|
+
check -x "${CLAUDE_DIR}/hooks/luciazero-statusline.sh" "hook luciazero-statusline.sh executable"
|
|
72
|
+
# stale hooks are the silent failure mode of `git pull && ./install.sh`
|
|
73
|
+
# without --with-hooks: sidecar updates, hook files do not
|
|
74
|
+
for HFILE in luciazero-verify.sh luciazero-statusline.sh; do
|
|
75
|
+
if cmp -s "${CLAUDE_DIR}/hooks/${HFILE}" "${SRC}/claude/hooks/${HFILE}"; then
|
|
76
|
+
echo " ok hooks/${HFILE} matches this checkout"
|
|
77
|
+
else
|
|
78
|
+
echo " MISS hooks/${HFILE} differs from this checkout (stale or customized) — re-run ./install.sh --with-hooks"; STATUS_RC=1
|
|
79
|
+
fi
|
|
80
|
+
done
|
|
81
|
+
WIRE_MISS=""
|
|
82
|
+
for SUB in edit bash stop session; do
|
|
83
|
+
grep -qF "${CLAUDE_DIR}/hooks/luciazero-verify.sh ${SUB}" "${CLAUDE_DIR}/settings.json" 2>/dev/null \
|
|
84
|
+
|| WIRE_MISS="${WIRE_MISS} ${SUB}"
|
|
85
|
+
done
|
|
86
|
+
if [ -z "${WIRE_MISS}" ]; then
|
|
87
|
+
echo " ok hooks wired in settings.json (edit/bash/stop/session)"
|
|
88
|
+
else
|
|
89
|
+
echo " MISS settings.json missing hook entries:${WIRE_MISS} (re-run ./install.sh --with-hooks)"; STATUS_RC=1
|
|
90
|
+
fi
|
|
91
|
+
if command -v python3 >/dev/null 2>&1; then
|
|
92
|
+
echo " ok python3 available (the hooks need it)"
|
|
93
|
+
else
|
|
94
|
+
# fail-open means a missing python3 breaks the hooks SILENTLY — surface it here
|
|
95
|
+
echo " MISS python3 not found — the installed hooks are failing open (doing nothing)"; STATUS_RC=1
|
|
96
|
+
fi
|
|
97
|
+
elif [ -f "${CLAUDE_DIR}/settings.json" ] \
|
|
98
|
+
&& grep -qF "${CLAUDE_DIR}/hooks/luciazero-" "${CLAUDE_DIR}/settings.json"; then
|
|
99
|
+
# worse than not installed: Claude Code keeps executing references to
|
|
100
|
+
# files that are gone — exactly what uninstall.sh works to prevent
|
|
101
|
+
echo " MISS settings.json references hook files that do not exist (dangling — re-run ./install.sh --with-hooks or ./uninstall.sh)"; STATUS_RC=1
|
|
102
|
+
else
|
|
103
|
+
echo " -- enforcement pack not installed (optional: ./install.sh --with-hooks)"
|
|
104
|
+
fi
|
|
105
|
+
exit "${STATUS_RC}"
|
|
106
|
+
fi
|
|
107
|
+
|
|
108
|
+
# collision-proof backup path for $1 (two runs in the same second must not overwrite)
|
|
109
|
+
bakpath() {
|
|
110
|
+
B="$1.bak.$(date +%Y%m%d%H%M%S)"
|
|
111
|
+
N=1
|
|
112
|
+
while [ -e "${B}" ]; do B="$1.bak.$(date +%Y%m%d%H%M%S).${N}"; N=$((N+1)); done
|
|
113
|
+
printf '%s' "${B}"
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
echo "Installing into ${CLAUDE_DIR}"
|
|
117
|
+
mkdir -p "${CLAUDE_DIR}/skills"
|
|
118
|
+
|
|
119
|
+
# 1. doctrine
|
|
120
|
+
cp "${SRC}/claude/${DOCTRINE}" "${CLAUDE_DIR}/${DOCTRINE}"
|
|
121
|
+
echo " ok ${DOCTRINE}"
|
|
122
|
+
|
|
123
|
+
# 2. skills
|
|
124
|
+
for SKILL in luciazero-bootstrap retro debug 'done' handoff experiment; do
|
|
125
|
+
rm -rf "${CLAUDE_DIR}/skills/${SKILL}"
|
|
126
|
+
cp -r "${SRC}/skills/${SKILL}" "${CLAUDE_DIR}/skills/${SKILL}"
|
|
127
|
+
echo " ok skills/${SKILL}"
|
|
128
|
+
done
|
|
129
|
+
|
|
130
|
+
# 3. reviewer agent (back up a pre-existing customized copy before overwriting)
|
|
131
|
+
mkdir -p "${CLAUDE_DIR}/agents"
|
|
132
|
+
AGENT="${CLAUDE_DIR}/agents/reviewer.md"
|
|
133
|
+
if [ -f "${AGENT}" ] && ! cmp -s "${SRC}/claude/agents/reviewer.md" "${AGENT}"; then
|
|
134
|
+
cp "${AGENT}" "$(bakpath "${AGENT}")"
|
|
135
|
+
echo " ok backed up existing agents/reviewer.md"
|
|
136
|
+
fi
|
|
137
|
+
cp "${SRC}/claude/agents/reviewer.md" "${AGENT}"
|
|
138
|
+
echo " ok agents/reviewer.md"
|
|
139
|
+
|
|
140
|
+
# 4. version sidecar — lets --status and future installs tell what is installed
|
|
141
|
+
V_NEW="$(version_of)"
|
|
142
|
+
V_OLD="$(cat "${CLAUDE_DIR}/.luciazero-version" 2>/dev/null || true)"
|
|
143
|
+
if [ -n "${V_NEW}" ]; then
|
|
144
|
+
if [ -n "${V_OLD}" ] && [ "${V_OLD}" != "${V_NEW}" ]; then
|
|
145
|
+
echo " ok updating ${V_OLD} -> ${V_NEW}"
|
|
146
|
+
fi
|
|
147
|
+
printf '%s\n' "${V_NEW}" > "${CLAUDE_DIR}/.luciazero-version"
|
|
148
|
+
fi
|
|
149
|
+
|
|
150
|
+
# 5. import line in global CLAUDE.md
|
|
151
|
+
GLOBAL_MD="${CLAUDE_DIR}/CLAUDE.md"
|
|
152
|
+
if [ -f "${GLOBAL_MD}" ] && grep -qF "${IMPORT_LINE}" "${GLOBAL_MD}"; then
|
|
153
|
+
echo " ok CLAUDE.md already imports ${DOCTRINE}"
|
|
154
|
+
else
|
|
155
|
+
if [ -f "${GLOBAL_MD}" ]; then
|
|
156
|
+
BACKUP="$(bakpath "${GLOBAL_MD}")"
|
|
157
|
+
cp "${GLOBAL_MD}" "${BACKUP}"
|
|
158
|
+
echo " ok backed up CLAUDE.md -> $(basename "${BACKUP}")"
|
|
159
|
+
printf '\n%s\n' "${IMPORT_LINE}" >> "${GLOBAL_MD}"
|
|
160
|
+
else
|
|
161
|
+
printf '%s\n' "${IMPORT_LINE}" > "${GLOBAL_MD}"
|
|
162
|
+
fi
|
|
163
|
+
echo " ok CLAUDE.md imports ${DOCTRINE}"
|
|
164
|
+
fi
|
|
165
|
+
|
|
166
|
+
# 6. enforcement pack (opt-in): hooks + statusline wired into settings.json
|
|
167
|
+
if [ "${WITH_HOOKS}" = 1 ]; then
|
|
168
|
+
command -v python3 >/dev/null 2>&1 || { echo "FAIL: --with-hooks requires python3" >&2; exit 1; }
|
|
169
|
+
mkdir -p "${CLAUDE_DIR}/hooks"
|
|
170
|
+
for H in luciazero-verify.sh luciazero-statusline.sh; do
|
|
171
|
+
DST="${CLAUDE_DIR}/hooks/${H}"
|
|
172
|
+
if [ -f "${DST}" ] && ! cmp -s "${SRC}/claude/hooks/${H}" "${DST}"; then
|
|
173
|
+
cp "${DST}" "$(bakpath "${DST}")"
|
|
174
|
+
echo " ok backed up existing hooks/${H}"
|
|
175
|
+
fi
|
|
176
|
+
cp "${SRC}/claude/hooks/${H}" "${DST}"
|
|
177
|
+
chmod +x "${DST}"
|
|
178
|
+
done
|
|
179
|
+
SETTINGS="${CLAUDE_DIR}/settings.json"
|
|
180
|
+
if [ -f "${SETTINGS}" ]; then
|
|
181
|
+
cp "${SETTINGS}" "$(bakpath "${SETTINGS}")"
|
|
182
|
+
fi
|
|
183
|
+
python3 - "${SETTINGS}" "${CLAUDE_DIR}/hooks" <<'PY' || { echo "FAIL: could not update settings.json (invalid JSON?) — hook files copied but not wired" >&2; exit 1; }
|
|
184
|
+
import json, os, sys
|
|
185
|
+
|
|
186
|
+
path, hooks_dir = sys.argv[1], sys.argv[2]
|
|
187
|
+
verify_cmd = os.path.join(hooks_dir, "luciazero-verify.sh")
|
|
188
|
+
status_cmd = os.path.join(hooks_dir, "luciazero-statusline.sh")
|
|
189
|
+
|
|
190
|
+
settings = {}
|
|
191
|
+
if os.path.exists(path):
|
|
192
|
+
with open(path) as f:
|
|
193
|
+
settings = json.load(f)
|
|
194
|
+
|
|
195
|
+
changed = False
|
|
196
|
+
hooks = settings.setdefault("hooks", {})
|
|
197
|
+
|
|
198
|
+
def ensure(event, matcher, command):
|
|
199
|
+
global changed
|
|
200
|
+
entries = hooks.setdefault(event, [])
|
|
201
|
+
for e in entries:
|
|
202
|
+
for h in e.get("hooks", []):
|
|
203
|
+
if h.get("command") == command:
|
|
204
|
+
return
|
|
205
|
+
entry = {"hooks": [{"type": "command", "command": command}]}
|
|
206
|
+
if matcher is not None:
|
|
207
|
+
entry["matcher"] = matcher
|
|
208
|
+
entries.append(entry)
|
|
209
|
+
changed = True
|
|
210
|
+
|
|
211
|
+
ensure("PostToolUse", "Edit|Write|NotebookEdit", verify_cmd + " edit")
|
|
212
|
+
ensure("PostToolUse", "Bash", verify_cmd + " bash")
|
|
213
|
+
ensure("Stop", None, verify_cmd + " stop")
|
|
214
|
+
ensure("SessionStart", None, verify_cmd + " session")
|
|
215
|
+
|
|
216
|
+
sl = settings.get("statusLine")
|
|
217
|
+
if sl is None:
|
|
218
|
+
settings["statusLine"] = {"type": "command", "command": status_cmd}
|
|
219
|
+
changed = True
|
|
220
|
+
print(" ok statusline wired")
|
|
221
|
+
elif sl.get("command") == status_cmd:
|
|
222
|
+
print(" ok statusline already wired")
|
|
223
|
+
else:
|
|
224
|
+
print(" !! statusline SKIPPED — a custom statusLine exists; to use ours, set")
|
|
225
|
+
print(" settings.json statusLine.command to: " + status_cmd)
|
|
226
|
+
|
|
227
|
+
if changed:
|
|
228
|
+
with open(path, "w") as f:
|
|
229
|
+
# ensure_ascii=False: an escaped non-ASCII config path (é) would
|
|
230
|
+
# never match --status's byte-level greps for the hook commands
|
|
231
|
+
json.dump(settings, f, indent=2, ensure_ascii=False)
|
|
232
|
+
f.write("\n")
|
|
233
|
+
print(" ok hooks wired into settings.json")
|
|
234
|
+
else:
|
|
235
|
+
print(" ok hooks already wired")
|
|
236
|
+
PY
|
|
237
|
+
fi
|
|
238
|
+
|
|
239
|
+
echo
|
|
240
|
+
echo "Done. Verify:"
|
|
241
|
+
echo " ./install.sh --status"
|
|
242
|
+
echo
|
|
243
|
+
echo "Skills: /luciazero-bootstrap, /debug, /done, /handoff, /experiment, /retro. Reviewer agent: 'reviewer'."
|
|
244
|
+
if [ "${WITH_HOOKS}" = 1 ]; then
|
|
245
|
+
echo "Enforcement pack installed: verify-tracking hooks + statusline (see settings.json)."
|
|
246
|
+
else
|
|
247
|
+
echo "Optional: ./install.sh --with-hooks adds the verify-nudge hooks + statusline."
|
|
248
|
+
fi
|
|
249
|
+
echo "The doctrine applies from the next Claude Code session."
|
package/package.json
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "luciazero",
|
|
3
|
+
"version": "1.5.0",
|
|
4
|
+
"description": "Verification-first discipline for coding agents (Claude Code + Codex CLI): 9-rule doctrine, six skills, reviewer agent, fail-open enforcement hooks. npx luciazero installs it.",
|
|
5
|
+
"repository": { "type": "git", "url": "git+https://github.com/ohm41321/luciazero.git" },
|
|
6
|
+
"homepage": "https://github.com/ohm41321/luciazero#readme",
|
|
7
|
+
"bugs": { "url": "https://github.com/ohm41321/luciazero/issues" },
|
|
8
|
+
"bin": {
|
|
9
|
+
"luciazero": "bin/luciazero.js"
|
|
10
|
+
},
|
|
11
|
+
"files": [
|
|
12
|
+
"bin",
|
|
13
|
+
"claude",
|
|
14
|
+
"skills",
|
|
15
|
+
"install.sh",
|
|
16
|
+
"uninstall.sh",
|
|
17
|
+
"install-codex.sh",
|
|
18
|
+
"uninstall-codex.sh",
|
|
19
|
+
"CHANGELOG.md"
|
|
20
|
+
],
|
|
21
|
+
"engines": {
|
|
22
|
+
"node": ">=18"
|
|
23
|
+
},
|
|
24
|
+
"license": "MIT",
|
|
25
|
+
"keywords": [
|
|
26
|
+
"claude-code",
|
|
27
|
+
"codex",
|
|
28
|
+
"agent-skills",
|
|
29
|
+
"verification",
|
|
30
|
+
"doctrine",
|
|
31
|
+
"hooks",
|
|
32
|
+
"agentic-engineering"
|
|
33
|
+
]
|
|
34
|
+
}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: debug
|
|
3
|
+
description: Hypothesis-driven debugging procedure. Use when a bug is not yet reliably reproduced, when a fix attempt just failed, when debugging has gone two or more iterations without progress, or when the user asks "why is this failing", "debug this properly", "ไล่บั๊ก". Not for trivial errors whose cause is already visible in the message.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Debug — hypothesis before edit
|
|
7
|
+
|
|
8
|
+
The doctrine says: *debugging starts with a hypothesis, not an edit.* Mutating code until the test goes green is not debugging — it is how plausible-but-wrong fixes ship. This is the procedure for bugs that resist the first obvious look.
|
|
9
|
+
|
|
10
|
+
## 1. Reproduce first
|
|
11
|
+
|
|
12
|
+
One command that shows the failure deterministically. This command is the ground truth for the whole session — every hypothesis is judged against it.
|
|
13
|
+
|
|
14
|
+
- If it cannot be reproduced yet, that is the entire current task. Do not theorize about causes of a failure you cannot trigger.
|
|
15
|
+
- If it is intermittent, make it deterministic before proceeding: fix the seed, pin the time/timezone, run it in a loop (`for i in $(seq 20)`) until the trigger condition is understood. An intermittent repro means the hypothesis space still contains "timing/state you have not seen".
|
|
16
|
+
|
|
17
|
+
## 2. Minimize
|
|
18
|
+
|
|
19
|
+
Shrink the reproduction — smaller input, fewer flags, one test instead of the suite — until the failure is small enough to reason about. Every element removed eliminates a family of hypotheses for free. Stop minimizing when shrinking stops being cheap.
|
|
20
|
+
|
|
21
|
+
## 3. Hypothesis ledger
|
|
22
|
+
|
|
23
|
+
**Seed it from recorded experience first.** Before inventing hypotheses, grep the symptom's keywords (error strings, subsystem names) against two files, if they exist:
|
|
24
|
+
|
|
25
|
+
- the repo's lesson ledger `docs/lessons.md` — this project's previously debugged failures;
|
|
26
|
+
- the global heuristics file `luciazero-heuristics.md` in the harness config dir (`~/.claude` / `~/.codex`) — cross-repo lessons.
|
|
27
|
+
|
|
28
|
+
A match becomes **H1** — still verify it with its `proven-by` command; a ledger match is a hypothesis with a head start, not a conclusion. No match, or no files: proceed normally.
|
|
29
|
+
|
|
30
|
+
Keep a visible ledger in the conversation. Each entry:
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
H<N>: <suspected cause> — refutable by: <command / observation> → <result: refuted | confirmed | pending>
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
- **Run the observation, not the edit.** Choose the cheapest command whose output discriminates between this hypothesis and the alternatives — a log line, a targeted print, a debugger break, one `grep`, `git bisect run <verify-cmd>` when a known-good commit exists.
|
|
37
|
+
- Prefer reading real state over reasoning about imagined state. The bug exists precisely because the mental model and reality differ — trust output.
|
|
38
|
+
- Dead hypotheses stay in the ledger marked refuted, so they are not silently retried an hour later.
|
|
39
|
+
|
|
40
|
+
## 4. One variable per iteration
|
|
41
|
+
|
|
42
|
+
- Change one thing, re-run the reproduction, record the result in the ledger.
|
|
43
|
+
- A fix attempt that failed gets **reverted before the next attempt** — stacked failed fixes create a second bug on top of the first.
|
|
44
|
+
- Two consecutive failed fixes on the same hypothesis means the hypothesis is dead, not unlucky. Widen the search: environment, dependency versions, input data, concurrency, or the test itself being wrong.
|
|
45
|
+
|
|
46
|
+
## 5. Close out
|
|
47
|
+
|
|
48
|
+
- The reproduction becomes a committed regression test: **red before the fix, green after** — run it both ways and quote both results. This proves the fix touched the actual cause. The mechanical form lives in the done skill's `scripts/` dir — run its `scripts/revert-probe.sh "<verify-cmd>"` from wherever that skill is installed (classic: `~/.claude/skills/done/`; plugin and `npx skills` installs keep it next to the done SKILL.md).
|
|
49
|
+
- Remove all instrumentation (prints, sleeps, debug flags) — check the diff for it explicitly.
|
|
50
|
+
- Run the full verify tier, not just the one test.
|
|
51
|
+
- If the session surfaced something reading the code cannot teach (a footgun, an environment quirk, a disproven approach), run `/retro` so the next session does not pay for this one's dead ends — for a debugged failure specifically, `/retro` records it in `docs/lessons.md` (symptom → cause → proven-by → fix), which is exactly what step 3 reads next time.
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: done
|
|
3
|
+
description: Closeout ritual before declaring any non-trivial task complete. Use when about to say "done", "finished", "it works now", before opening a PR, when the user asks "is it done?", "wrap it up", "ปิดงาน" — or whenever a change is about to be handed back as complete. Not for trivial single-line answers with no code change.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Done — prove it before you say it
|
|
7
|
+
|
|
8
|
+
The doctrine says: *done is proven by a command, not by my judgment.* This is the ritual that turns that rule into a checklist. Run every step; skipping one is how "done" ships broken.
|
|
9
|
+
|
|
10
|
+
## 1. Full verify
|
|
11
|
+
|
|
12
|
+
Run the **full** tier (`verify-full` if the repo has two tiers, else the verify command). Quote the shortest decisive line of real output.
|
|
13
|
+
|
|
14
|
+
- Red → you are not here yet. Go back to the loop; do not continue this ritual.
|
|
15
|
+
- No verify command exists → that is the first bug (`/luciazero-bootstrap`). Say so instead of declaring done.
|
|
16
|
+
- The command must actually have run **now**, in this session — a green from an hour ago proves the past, not the present.
|
|
17
|
+
|
|
18
|
+
## 2. Skeptic diff pass
|
|
19
|
+
|
|
20
|
+
Re-read the final diff as a hostile reviewer. Tests prove what they cover; hunt what they do not:
|
|
21
|
+
|
|
22
|
+
- **Edge cases** — empty, zero, negative, unicode, first/last, concurrent
|
|
23
|
+
- **Error paths** — the call fails, the file is missing, the network drops; are errors swallowed?
|
|
24
|
+
- **Changed contracts** — public API shape, serialized formats, schema, config keys: who consumes the old shape?
|
|
25
|
+
- **Accidental content** — files touched by mistake, debug prints, commented-out code, leftover instrumentation, loosened dependency pins, secrets
|
|
26
|
+
- **Test honesty** — would the new/changed tests fail if the change were reverted? The mechanical form: `scripts/revert-probe.sh "<verify-cmd>"` answers it in one command. Weakened or deleted checks are findings, not cleanup.
|
|
27
|
+
|
|
28
|
+
Fix what you find, re-run step 1, then continue.
|
|
29
|
+
|
|
30
|
+
## 3. Independent review, if the diff earns it
|
|
31
|
+
|
|
32
|
+
Get an adversarial second opinion — the harness's built-in review command (Claude Code: `/code-review`) or the `reviewer` agent — when **any** of these hold:
|
|
33
|
+
|
|
34
|
+
- Touches a public API, data migration, auth, money, or concurrency
|
|
35
|
+
- Wide diff (many files, or a subsystem you did not previously know)
|
|
36
|
+
- You cannot explain in one sentence why each hunk is safe
|
|
37
|
+
|
|
38
|
+
Findings go back through step 1. For small well-understood diffs, step 2 suffices — do not add ceremony the diff does not need.
|
|
39
|
+
|
|
40
|
+
## 4. Scope check
|
|
41
|
+
|
|
42
|
+
Re-read the original request. For each part: delivered, or named as left out with the reason. Silently dropped scope is the failure mode this step exists to catch. Anything left out gets said **plainly** in the report, not buried.
|
|
43
|
+
|
|
44
|
+
## 5. Lessons
|
|
45
|
+
|
|
46
|
+
If the session hit a dead end, a footgun, or disproved a tempting approach — run `/retro` now, while the evidence is fresh. If unfinished work remains for a future session, `/handoff` instead.
|
|
47
|
+
|
|
48
|
+
## 6. Report
|
|
49
|
+
|
|
50
|
+
```
|
|
51
|
+
Done: <what changed, one line>
|
|
52
|
+
Proof: <verify command> → <decisive output line>
|
|
53
|
+
Not covered: <what verify does not prove>
|
|
54
|
+
Left out: <scope not delivered + why, or "nothing">
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
No hedging in the report: if all steps passed, state it plainly; if one did not, the task is not done and the report says what remains instead.
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# revert-probe.sh — the mechanical form of the done-skill question "would the
|
|
3
|
+
# new tests fail if the change were reverted?" (doctrine: red before green).
|
|
4
|
+
# Checks <base-ref> out into a throwaway worktree, copies ONLY the test files
|
|
5
|
+
# changed since <base-ref> from the working tree on top of it, and runs the
|
|
6
|
+
# verify command there. The result is INVERTED: old code failing the new
|
|
7
|
+
# tests is the PASS.
|
|
8
|
+
#
|
|
9
|
+
# Usage: revert-probe.sh "<verify-cmd>" [base-ref] (base-ref default: HEAD)
|
|
10
|
+
# Run it BEFORE committing — the fix and its new tests sit in the working
|
|
11
|
+
# tree while HEAD is still the old code. For an already-committed fix, pass
|
|
12
|
+
# the pre-fix ref (e.g. HEAD~1) as base-ref.
|
|
13
|
+
#
|
|
14
|
+
# Exit: 0 tests bite · 1 tests stay green on old code, or no changed test
|
|
15
|
+
# files · 2 UNASSESSABLE (not a git repo, no commits, invalid base).
|
|
16
|
+
# Pure bash + git; never touches the caller's working tree.
|
|
17
|
+
set -euo pipefail
|
|
18
|
+
|
|
19
|
+
VERIFY="${1:?usage: revert-probe.sh \"<verify-cmd>\" [base-ref]}"
|
|
20
|
+
BASE="${2:-HEAD}"
|
|
21
|
+
|
|
22
|
+
unassessable() { echo "UNASSESSABLE: $*"; exit 2; }
|
|
23
|
+
|
|
24
|
+
git rev-parse --git-dir >/dev/null 2>&1 || unassessable "not a git repo"
|
|
25
|
+
git rev-parse --verify HEAD >/dev/null 2>&1 || unassessable "no commits yet"
|
|
26
|
+
git rev-parse --verify --quiet "${BASE}^{commit}" >/dev/null 2>&1 \
|
|
27
|
+
|| unassessable "invalid base ref: ${BASE}"
|
|
28
|
+
TOP="$(git rev-parse --show-toplevel 2>/dev/null)" || unassessable "no working tree (bare repo?)"
|
|
29
|
+
cd "${TOP}"
|
|
30
|
+
|
|
31
|
+
# test-file patterns mirror luciazero-bootstrap's detect.sh: tests-style dirs
|
|
32
|
+
# plus test_*.*, *_test.*, *.test.*, *.spec.* file names
|
|
33
|
+
is_test_file() {
|
|
34
|
+
case "/$1" in
|
|
35
|
+
*/tests/*|*/test/*|*/spec/*|*/__tests__/*) return 0 ;;
|
|
36
|
+
esac
|
|
37
|
+
case "${1##*/}" in
|
|
38
|
+
test_*.*|*_test.*|*.test.*|*.spec.*) return 0 ;;
|
|
39
|
+
esac
|
|
40
|
+
return 1
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
# scratch space first — the changed-file list is stored NUL-delimited in a
|
|
44
|
+
# file, because git C-quotes non-ASCII/backslash names in its plain output
|
|
45
|
+
# (-z emits them raw) and bash variables cannot hold NUL bytes
|
|
46
|
+
TMP="$(mktemp -d)"
|
|
47
|
+
WT="${TMP}/worktree"
|
|
48
|
+
trap 'git worktree remove --force "${WT}" >/dev/null 2>&1 || true
|
|
49
|
+
rm -rf "${TMP}"
|
|
50
|
+
git worktree prune >/dev/null 2>&1 || true' EXIT
|
|
51
|
+
|
|
52
|
+
# changed vs base (tracked) plus untracked — the two sets are disjoint —
|
|
53
|
+
# filtered to test files that still exist (a deleted test cannot bite)
|
|
54
|
+
TEST_LIST="${TMP}/tests"
|
|
55
|
+
: > "${TEST_LIST}"
|
|
56
|
+
COUNT=0
|
|
57
|
+
while IFS= read -r -d '' F; do
|
|
58
|
+
[ -f "${F}" ] || continue
|
|
59
|
+
if is_test_file "${F}"; then
|
|
60
|
+
printf '%s\0' "${F}" >> "${TEST_LIST}"
|
|
61
|
+
COUNT=$((COUNT + 1))
|
|
62
|
+
fi
|
|
63
|
+
done < <({ git diff --name-only -z "${BASE}" --; git ls-files --others --exclude-standard -z; } 2>/dev/null)
|
|
64
|
+
|
|
65
|
+
if [ "${COUNT}" -eq 0 ]; then
|
|
66
|
+
echo "FAIL: no test files changed since ${BASE} — the change ships without a test that bites"
|
|
67
|
+
exit 1
|
|
68
|
+
fi
|
|
69
|
+
|
|
70
|
+
# old code in a throwaway worktree; cleanup runs on every exit path
|
|
71
|
+
git worktree add --detach "${WT}" "${BASE}" >/dev/null 2>&1 \
|
|
72
|
+
|| unassessable "git worktree add failed for ${BASE}"
|
|
73
|
+
|
|
74
|
+
# overlay ONLY the changed test files from the working tree
|
|
75
|
+
while IFS= read -r -d '' F; do
|
|
76
|
+
mkdir -p "${WT}/$(dirname "${F}")"
|
|
77
|
+
cp "${F}" "${WT}/${F}"
|
|
78
|
+
done < "${TEST_LIST}"
|
|
79
|
+
|
|
80
|
+
RC=0
|
|
81
|
+
OUT="$(cd "${WT}" && sh -c "${VERIFY}" 2>&1)" || RC=$?
|
|
82
|
+
|
|
83
|
+
if [ "${RC}" -ne 0 ]; then
|
|
84
|
+
LAST=""
|
|
85
|
+
while IFS= read -r LINE; do
|
|
86
|
+
case "${LINE}" in *[![:space:]]*) LAST="${LINE}" ;; esac
|
|
87
|
+
done <<EOF
|
|
88
|
+
${OUT}
|
|
89
|
+
EOF
|
|
90
|
+
echo "PASS: regression tests bite — old code fails the new tests"
|
|
91
|
+
echo " evidence (exit ${RC}): ${LAST:-<no output>}"
|
|
92
|
+
exit 0
|
|
93
|
+
fi
|
|
94
|
+
echo "FAIL: the changed tests stay green against the old code — they do not cover the change"
|
|
95
|
+
exit 1
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: experiment
|
|
3
|
+
description: Measured-change protocol for performance and tuning work. Use when the task is "make it faster", "optimize", "reduce memory", "ทดลอง", when comparing two approaches, or whenever a claim like "this should be faster" is about to be made without a number. Not for correctness bugs — that is /debug.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Experiment — no claim without a measurement
|
|
7
|
+
|
|
8
|
+
An optimization without a baseline is a guess with confidence. The protocol is the same loop as always — but *verify* here means **measure**, and the doctrine's "never re-derive a dead end twice" means null results get recorded with the same weight as wins.
|
|
9
|
+
|
|
10
|
+
## 1. Define the metric before touching code
|
|
11
|
+
|
|
12
|
+
- One command that prints the number: runtime, RSS, p95 latency, binary size, query count. If no such command exists, building it is step zero (the measurement twin of "no verify command is the first bug").
|
|
13
|
+
- Decide **now** what improvement would count — "worth it if ≥10% faster" — so the verdict is not negotiated after the numbers exist.
|
|
14
|
+
|
|
15
|
+
## 2. Baseline
|
|
16
|
+
|
|
17
|
+
- Run the measurement **at least 3 times**; record all values, not just the mean — the spread is what separates signal from noise.
|
|
18
|
+
- Pin what you can: fixed seed, same input data, warm/cold state chosen deliberately, machine as quiet as you can get it. Note what you could not pin.
|
|
19
|
+
- Correctness verify must be green before and after — a fast wrong answer is not an optimization.
|
|
20
|
+
|
|
21
|
+
## 3. One variable per experiment
|
|
22
|
+
|
|
23
|
+
Change one thing. Two changes in one measurement produce a number that explains neither. (Same discipline as `/debug`; same reason.)
|
|
24
|
+
|
|
25
|
+
## 4. Measure again
|
|
26
|
+
|
|
27
|
+
- Same command, same repetitions, same conditions.
|
|
28
|
+
- The difference counts only if it clearly beats the baseline spread. Inside the noise = **null result**, not "slightly faster".
|
|
29
|
+
|
|
30
|
+
## 5. Verdict and record
|
|
31
|
+
|
|
32
|
+
Append to `docs/experiments.md` (create it if absent; follow the repo's existing log if one exists):
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
## <date> — <hypothesis, one line>
|
|
36
|
+
change: <what was changed, file/approach>
|
|
37
|
+
baseline: <values> | result: <values>
|
|
38
|
+
verdict: WIN <n%> | NULL (inside noise) | LOSS
|
|
39
|
+
decision: <kept / reverted> — <one-line reason>
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
- **Losers and nulls are reverted immediately** — the log keeps the knowledge, the tree keeps only wins.
|
|
43
|
+
- A null result is a finding: "tried X, no measurable gain — do not retry without new evidence" saves the next session the same hour. If it is load-bearing, surface it via `/retro` into the project notes too.
|
|
44
|
+
- Never delete a previous entry; if a new experiment overturns an old one, add the new entry and cross-reference.
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: handoff
|
|
3
|
+
description: Write a state capsule so the next session — or a different agent/harness — can resume unfinished work without re-deriving context. Use when a session is ending with work incomplete, when the user says "handoff", "pack up", "ส่งต่อ", "continue tomorrow", when switching between Claude Code and Codex mid-task, or when context is about to be compacted away on a long task.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Handoff — state that survives the session
|
|
7
|
+
|
|
8
|
+
`/retro` records **permanent lessons**; this records **transient state**. A good capsule lets a stranger (including future-you with zero context) type one command and be productive in two minutes. A stale capsule is worse than none — so capsules are consumed and deleted, never accumulated.
|
|
9
|
+
|
|
10
|
+
## 1. Write the capsule
|
|
11
|
+
|
|
12
|
+
Create `HANDOFF.md` at the repo root:
|
|
13
|
+
|
|
14
|
+
```markdown
|
|
15
|
+
# Handoff — <date>
|
|
16
|
+
|
|
17
|
+
## Goal
|
|
18
|
+
<the original request, one paragraph, verbatim enough to re-anchor>
|
|
19
|
+
|
|
20
|
+
## State
|
|
21
|
+
- Done: <what is finished, and the verify evidence: command → decisive line>
|
|
22
|
+
- In progress: <the exact piece mid-flight, and which files hold it>
|
|
23
|
+
- Verify: <the command(s) to run, fast and full tier>
|
|
24
|
+
|
|
25
|
+
## Next step
|
|
26
|
+
<ONE literal command or edit to do first — not a theme, an action>
|
|
27
|
+
|
|
28
|
+
## Open hypotheses
|
|
29
|
+
- H1: <suspected cause / approach> — status: <untested | supported by X | refuted by Y>
|
|
30
|
+
|
|
31
|
+
## Landmines
|
|
32
|
+
- <thing that looks safe but is not, discovered this session>
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
Rules:
|
|
36
|
+
|
|
37
|
+
- **The next step is literal.** "Continue the refactor" is not a next step; "run `pytest tests/test_auth.py -k refresh` — it is the failing one" is.
|
|
38
|
+
- **State only what this session knows.** No aspirations, no backlog — that belongs in the issue tracker.
|
|
39
|
+
- **Refuted hypotheses stay in the capsule** — they are exactly what the next session would otherwise waste an hour re-deriving. If a lesson is permanent (true beyond this task), it also goes through `/retro`.
|
|
40
|
+
- Uncommitted changes: say so explicitly, and name the files. The capsule must not imply a clean tree that is not clean.
|
|
41
|
+
|
|
42
|
+
## 2. Route it
|
|
43
|
+
|
|
44
|
+
- **Same machine, next session** — `HANDOFF.md` in the repo root is enough. Do not commit it.
|
|
45
|
+
- **Cross-machine or another person/agent** — commit it on the working branch (it travels with the code), and say in the final message that it exists.
|
|
46
|
+
- The project's notes file does **not** get the capsule — notes are permanent, capsules are transient. A one-line pointer is fine if the repo's convention wants one.
|
|
47
|
+
|
|
48
|
+
## 3. Consume protocol (for the reader)
|
|
49
|
+
|
|
50
|
+
A session that finds `HANDOFF.md`:
|
|
51
|
+
|
|
52
|
+
1. Read it **before** touching the code.
|
|
53
|
+
2. Run the Verify command(s) to confirm the described state is still true — the capsule describes the past; the tree is the truth.
|
|
54
|
+
3. **Delete the capsule** (or `git rm` on the branch) once absorbed. Never leave a consumed capsule to go stale; never update one incrementally across many sessions — write a fresh one at each handoff.
|
|
55
|
+
|
|
56
|
+
If the capsule and the tree disagree, trust the tree and say so.
|