luciazero 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/install.sh ADDED
@@ -0,0 +1,249 @@
1
+ #!/usr/bin/env bash
2
+ # Install the Luciazero doctrine + skills into ~/.claude/
3
+ # Idempotent. Backs up CLAUDE.md (and settings.json when --with-hooks)
4
+ # before editing. Writes nothing outside ~/.claude/.
5
+ #
6
+ # ./install.sh doctrine + skills + reviewer agent
7
+ # ./install.sh --with-hooks also wire the enforcement pack: verify-tracking
8
+ # hooks + statusline into ~/.claude/settings.json
9
+ # (Claude Code only; requires python3)
10
+ # ./install.sh --status read-only health check of an existing install;
11
+ # exits non-zero if a core piece is missing
12
+ set -euo pipefail
13
+
14
+ WITH_HOOKS=0
15
+ STATUS_ONLY=0
16
+ for ARG in "$@"; do
17
+ case "${ARG}" in
18
+ --with-hooks) WITH_HOOKS=1 ;;
19
+ --status) STATUS_ONLY=1 ;;
20
+ *) echo "unknown option: ${ARG} (supported: --with-hooks, --status)" >&2; exit 1 ;;
21
+ esac
22
+ done
23
+
24
+ SRC="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
25
+ CLAUDE_DIR="${CLAUDE_CONFIG_DIR:-$HOME/.claude}"
26
+ DOCTRINE="luciazero.md"
27
+ IMPORT_LINE="@${DOCTRINE}"
28
+
29
+ # newest released version in this checkout's CHANGELOG (informational)
30
+ version_of() {
31
+ grep -m1 -oE '^## \[[0-9]+\.[0-9]+\.[0-9]+\]' "${SRC}/CHANGELOG.md" 2>/dev/null | tr -d '#[] ' || true
32
+ }
33
+
34
+ if [ "${STATUS_ONLY}" = 1 ]; then
35
+ echo "Status of ${CLAUDE_DIR} (read-only)"
36
+ STATUS_RC=0
37
+ check() { # check <file-test-flag: -f|-x> <path> <label...>
38
+ T="$1"; P="$2"; shift 2
39
+ OK=0
40
+ case "${T}" in
41
+ -x) if [ -x "${P}" ]; then OK=1; fi ;;
42
+ *) if [ -f "${P}" ]; then OK=1; fi ;;
43
+ esac
44
+ if [ "${OK}" = 1 ]; then echo " ok $*"; else echo " MISS $*"; STATUS_RC=1; fi
45
+ }
46
+ check -f "${CLAUDE_DIR}/${DOCTRINE}" "doctrine ${DOCTRINE}"
47
+ for SKILL in luciazero-bootstrap retro debug 'done' handoff experiment; do
48
+ check -f "${CLAUDE_DIR}/skills/${SKILL}/SKILL.md" "skill ${SKILL}"
49
+ done
50
+ check -x "${CLAUDE_DIR}/skills/luciazero-bootstrap/scripts/detect.sh" "detect.sh executable"
51
+ check -x "${CLAUDE_DIR}/skills/done/scripts/revert-probe.sh" "revert-probe.sh executable"
52
+ check -f "${CLAUDE_DIR}/agents/reviewer.md" "reviewer agent"
53
+ GLOBAL_MD="${CLAUDE_DIR}/CLAUDE.md"
54
+ N="$(grep -cxF "${IMPORT_LINE}" "${GLOBAL_MD}" 2>/dev/null || true)"
55
+ if [ "${N:-0}" = 1 ]; then
56
+ echo " ok CLAUDE.md imports the doctrine"
57
+ else
58
+ echo " MISS CLAUDE.md import line (${IMPORT_LINE} exactly once; found ${N:-0})"; STATUS_RC=1
59
+ fi
60
+ V_SRC="$(version_of)"
61
+ V_INST="$(cat "${CLAUDE_DIR}/.luciazero-version" 2>/dev/null || true)"
62
+ if [ -z "${V_INST}" ]; then
63
+ echo " -- installed version unknown (no sidecar — installed by an older version)"
64
+ elif [ "${V_INST}" = "${V_SRC}" ]; then
65
+ echo " ok version ${V_INST} (matches this checkout)"
66
+ else
67
+ echo " !! installed ${V_INST}, checkout ${V_SRC:-?} — re-run ./install.sh to update"
68
+ fi
69
+ if [ -f "${CLAUDE_DIR}/hooks/luciazero-verify.sh" ]; then
70
+ check -x "${CLAUDE_DIR}/hooks/luciazero-verify.sh" "hook luciazero-verify.sh executable"
71
+ check -x "${CLAUDE_DIR}/hooks/luciazero-statusline.sh" "hook luciazero-statusline.sh executable"
72
+ # stale hooks are the silent failure mode of `git pull && ./install.sh`
73
+ # without --with-hooks: sidecar updates, hook files do not
74
+ for HFILE in luciazero-verify.sh luciazero-statusline.sh; do
75
+ if cmp -s "${CLAUDE_DIR}/hooks/${HFILE}" "${SRC}/claude/hooks/${HFILE}"; then
76
+ echo " ok hooks/${HFILE} matches this checkout"
77
+ else
78
+ echo " MISS hooks/${HFILE} differs from this checkout (stale or customized) — re-run ./install.sh --with-hooks"; STATUS_RC=1
79
+ fi
80
+ done
81
+ WIRE_MISS=""
82
+ for SUB in edit bash stop session; do
83
+ grep -qF "${CLAUDE_DIR}/hooks/luciazero-verify.sh ${SUB}" "${CLAUDE_DIR}/settings.json" 2>/dev/null \
84
+ || WIRE_MISS="${WIRE_MISS} ${SUB}"
85
+ done
86
+ if [ -z "${WIRE_MISS}" ]; then
87
+ echo " ok hooks wired in settings.json (edit/bash/stop/session)"
88
+ else
89
+ echo " MISS settings.json missing hook entries:${WIRE_MISS} (re-run ./install.sh --with-hooks)"; STATUS_RC=1
90
+ fi
91
+ if command -v python3 >/dev/null 2>&1; then
92
+ echo " ok python3 available (the hooks need it)"
93
+ else
94
+ # fail-open means a missing python3 breaks the hooks SILENTLY — surface it here
95
+ echo " MISS python3 not found — the installed hooks are failing open (doing nothing)"; STATUS_RC=1
96
+ fi
97
+ elif [ -f "${CLAUDE_DIR}/settings.json" ] \
98
+ && grep -qF "${CLAUDE_DIR}/hooks/luciazero-" "${CLAUDE_DIR}/settings.json"; then
99
+ # worse than not installed: Claude Code keeps executing references to
100
+ # files that are gone — exactly what uninstall.sh works to prevent
101
+ echo " MISS settings.json references hook files that do not exist (dangling — re-run ./install.sh --with-hooks or ./uninstall.sh)"; STATUS_RC=1
102
+ else
103
+ echo " -- enforcement pack not installed (optional: ./install.sh --with-hooks)"
104
+ fi
105
+ exit "${STATUS_RC}"
106
+ fi
107
+
108
+ # collision-proof backup path for $1 (two runs in the same second must not overwrite)
109
+ bakpath() {
110
+ B="$1.bak.$(date +%Y%m%d%H%M%S)"
111
+ N=1
112
+ while [ -e "${B}" ]; do B="$1.bak.$(date +%Y%m%d%H%M%S).${N}"; N=$((N+1)); done
113
+ printf '%s' "${B}"
114
+ }
115
+
116
+ echo "Installing into ${CLAUDE_DIR}"
117
+ mkdir -p "${CLAUDE_DIR}/skills"
118
+
119
+ # 1. doctrine
120
+ cp "${SRC}/claude/${DOCTRINE}" "${CLAUDE_DIR}/${DOCTRINE}"
121
+ echo " ok ${DOCTRINE}"
122
+
123
+ # 2. skills
124
+ for SKILL in luciazero-bootstrap retro debug 'done' handoff experiment; do
125
+ rm -rf "${CLAUDE_DIR}/skills/${SKILL}"
126
+ cp -r "${SRC}/skills/${SKILL}" "${CLAUDE_DIR}/skills/${SKILL}"
127
+ echo " ok skills/${SKILL}"
128
+ done
129
+
130
+ # 3. reviewer agent (back up a pre-existing customized copy before overwriting)
131
+ mkdir -p "${CLAUDE_DIR}/agents"
132
+ AGENT="${CLAUDE_DIR}/agents/reviewer.md"
133
+ if [ -f "${AGENT}" ] && ! cmp -s "${SRC}/claude/agents/reviewer.md" "${AGENT}"; then
134
+ cp "${AGENT}" "$(bakpath "${AGENT}")"
135
+ echo " ok backed up existing agents/reviewer.md"
136
+ fi
137
+ cp "${SRC}/claude/agents/reviewer.md" "${AGENT}"
138
+ echo " ok agents/reviewer.md"
139
+
140
+ # 4. version sidecar — lets --status and future installs tell what is installed
141
+ V_NEW="$(version_of)"
142
+ V_OLD="$(cat "${CLAUDE_DIR}/.luciazero-version" 2>/dev/null || true)"
143
+ if [ -n "${V_NEW}" ]; then
144
+ if [ -n "${V_OLD}" ] && [ "${V_OLD}" != "${V_NEW}" ]; then
145
+ echo " ok updating ${V_OLD} -> ${V_NEW}"
146
+ fi
147
+ printf '%s\n' "${V_NEW}" > "${CLAUDE_DIR}/.luciazero-version"
148
+ fi
149
+
150
+ # 5. import line in global CLAUDE.md
151
+ GLOBAL_MD="${CLAUDE_DIR}/CLAUDE.md"
152
+ if [ -f "${GLOBAL_MD}" ] && grep -qF "${IMPORT_LINE}" "${GLOBAL_MD}"; then
153
+ echo " ok CLAUDE.md already imports ${DOCTRINE}"
154
+ else
155
+ if [ -f "${GLOBAL_MD}" ]; then
156
+ BACKUP="$(bakpath "${GLOBAL_MD}")"
157
+ cp "${GLOBAL_MD}" "${BACKUP}"
158
+ echo " ok backed up CLAUDE.md -> $(basename "${BACKUP}")"
159
+ printf '\n%s\n' "${IMPORT_LINE}" >> "${GLOBAL_MD}"
160
+ else
161
+ printf '%s\n' "${IMPORT_LINE}" > "${GLOBAL_MD}"
162
+ fi
163
+ echo " ok CLAUDE.md imports ${DOCTRINE}"
164
+ fi
165
+
166
+ # 6. enforcement pack (opt-in): hooks + statusline wired into settings.json
167
+ if [ "${WITH_HOOKS}" = 1 ]; then
168
+ command -v python3 >/dev/null 2>&1 || { echo "FAIL: --with-hooks requires python3" >&2; exit 1; }
169
+ mkdir -p "${CLAUDE_DIR}/hooks"
170
+ for H in luciazero-verify.sh luciazero-statusline.sh; do
171
+ DST="${CLAUDE_DIR}/hooks/${H}"
172
+ if [ -f "${DST}" ] && ! cmp -s "${SRC}/claude/hooks/${H}" "${DST}"; then
173
+ cp "${DST}" "$(bakpath "${DST}")"
174
+ echo " ok backed up existing hooks/${H}"
175
+ fi
176
+ cp "${SRC}/claude/hooks/${H}" "${DST}"
177
+ chmod +x "${DST}"
178
+ done
179
+ SETTINGS="${CLAUDE_DIR}/settings.json"
180
+ if [ -f "${SETTINGS}" ]; then
181
+ cp "${SETTINGS}" "$(bakpath "${SETTINGS}")"
182
+ fi
183
+ python3 - "${SETTINGS}" "${CLAUDE_DIR}/hooks" <<'PY' || { echo "FAIL: could not update settings.json (invalid JSON?) — hook files copied but not wired" >&2; exit 1; }
184
+ import json, os, sys
185
+
186
+ path, hooks_dir = sys.argv[1], sys.argv[2]
187
+ verify_cmd = os.path.join(hooks_dir, "luciazero-verify.sh")
188
+ status_cmd = os.path.join(hooks_dir, "luciazero-statusline.sh")
189
+
190
+ settings = {}
191
+ if os.path.exists(path):
192
+ with open(path) as f:
193
+ settings = json.load(f)
194
+
195
+ changed = False
196
+ hooks = settings.setdefault("hooks", {})
197
+
198
+ def ensure(event, matcher, command):
199
+ global changed
200
+ entries = hooks.setdefault(event, [])
201
+ for e in entries:
202
+ for h in e.get("hooks", []):
203
+ if h.get("command") == command:
204
+ return
205
+ entry = {"hooks": [{"type": "command", "command": command}]}
206
+ if matcher is not None:
207
+ entry["matcher"] = matcher
208
+ entries.append(entry)
209
+ changed = True
210
+
211
+ ensure("PostToolUse", "Edit|Write|NotebookEdit", verify_cmd + " edit")
212
+ ensure("PostToolUse", "Bash", verify_cmd + " bash")
213
+ ensure("Stop", None, verify_cmd + " stop")
214
+ ensure("SessionStart", None, verify_cmd + " session")
215
+
216
+ sl = settings.get("statusLine")
217
+ if sl is None:
218
+ settings["statusLine"] = {"type": "command", "command": status_cmd}
219
+ changed = True
220
+ print(" ok statusline wired")
221
+ elif sl.get("command") == status_cmd:
222
+ print(" ok statusline already wired")
223
+ else:
224
+ print(" !! statusline SKIPPED — a custom statusLine exists; to use ours, set")
225
+ print(" settings.json statusLine.command to: " + status_cmd)
226
+
227
+ if changed:
228
+ with open(path, "w") as f:
229
+ # ensure_ascii=False: an escaped non-ASCII config path (é) would
230
+ # never match --status's byte-level greps for the hook commands
231
+ json.dump(settings, f, indent=2, ensure_ascii=False)
232
+ f.write("\n")
233
+ print(" ok hooks wired into settings.json")
234
+ else:
235
+ print(" ok hooks already wired")
236
+ PY
237
+ fi
238
+
239
+ echo
240
+ echo "Done. Verify:"
241
+ echo " ./install.sh --status"
242
+ echo
243
+ echo "Skills: /luciazero-bootstrap, /debug, /done, /handoff, /experiment, /retro. Reviewer agent: 'reviewer'."
244
+ if [ "${WITH_HOOKS}" = 1 ]; then
245
+ echo "Enforcement pack installed: verify-tracking hooks + statusline (see settings.json)."
246
+ else
247
+ echo "Optional: ./install.sh --with-hooks adds the verify-nudge hooks + statusline."
248
+ fi
249
+ echo "The doctrine applies from the next Claude Code session."
package/package.json ADDED
@@ -0,0 +1,34 @@
1
+ {
2
+ "name": "luciazero",
3
+ "version": "1.5.0",
4
+ "description": "Verification-first discipline for coding agents (Claude Code + Codex CLI): 9-rule doctrine, six skills, reviewer agent, fail-open enforcement hooks. npx luciazero installs it.",
5
+ "repository": { "type": "git", "url": "git+https://github.com/ohm41321/luciazero.git" },
6
+ "homepage": "https://github.com/ohm41321/luciazero#readme",
7
+ "bugs": { "url": "https://github.com/ohm41321/luciazero/issues" },
8
+ "bin": {
9
+ "luciazero": "bin/luciazero.js"
10
+ },
11
+ "files": [
12
+ "bin",
13
+ "claude",
14
+ "skills",
15
+ "install.sh",
16
+ "uninstall.sh",
17
+ "install-codex.sh",
18
+ "uninstall-codex.sh",
19
+ "CHANGELOG.md"
20
+ ],
21
+ "engines": {
22
+ "node": ">=18"
23
+ },
24
+ "license": "MIT",
25
+ "keywords": [
26
+ "claude-code",
27
+ "codex",
28
+ "agent-skills",
29
+ "verification",
30
+ "doctrine",
31
+ "hooks",
32
+ "agentic-engineering"
33
+ ]
34
+ }
@@ -0,0 +1,51 @@
1
+ ---
2
+ name: debug
3
+ description: Hypothesis-driven debugging procedure. Use when a bug is not yet reliably reproduced, when a fix attempt just failed, when debugging has gone two or more iterations without progress, or when the user asks "why is this failing", "debug this properly", "ไล่บั๊ก". Not for trivial errors whose cause is already visible in the message.
4
+ ---
5
+
6
+ # Debug — hypothesis before edit
7
+
8
+ The doctrine says: *debugging starts with a hypothesis, not an edit.* Mutating code until the test goes green is not debugging — it is how plausible-but-wrong fixes ship. This is the procedure for bugs that resist the first obvious look.
9
+
10
+ ## 1. Reproduce first
11
+
12
+ One command that shows the failure deterministically. This command is the ground truth for the whole session — every hypothesis is judged against it.
13
+
14
+ - If it cannot be reproduced yet, that is the entire current task. Do not theorize about causes of a failure you cannot trigger.
15
+ - If it is intermittent, make it deterministic before proceeding: fix the seed, pin the time/timezone, run it in a loop (`for i in $(seq 20)`) until the trigger condition is understood. An intermittent repro means the hypothesis space still contains "timing/state you have not seen".
16
+
17
+ ## 2. Minimize
18
+
19
+ Shrink the reproduction — smaller input, fewer flags, one test instead of the suite — until the failure is small enough to reason about. Every element removed eliminates a family of hypotheses for free. Stop minimizing when shrinking stops being cheap.
20
+
21
+ ## 3. Hypothesis ledger
22
+
23
+ **Seed it from recorded experience first.** Before inventing hypotheses, grep the symptom's keywords (error strings, subsystem names) against two files, if they exist:
24
+
25
+ - the repo's lesson ledger `docs/lessons.md` — this project's previously debugged failures;
26
+ - the global heuristics file `luciazero-heuristics.md` in the harness config dir (`~/.claude` / `~/.codex`) — cross-repo lessons.
27
+
28
+ A match becomes **H1** — still verify it with its `proven-by` command; a ledger match is a hypothesis with a head start, not a conclusion. No match, or no files: proceed normally.
29
+
30
+ Keep a visible ledger in the conversation. Each entry:
31
+
32
+ ```
33
+ H<N>: <suspected cause> — refutable by: <command / observation> → <result: refuted | confirmed | pending>
34
+ ```
35
+
36
+ - **Run the observation, not the edit.** Choose the cheapest command whose output discriminates between this hypothesis and the alternatives — a log line, a targeted print, a debugger break, one `grep`, `git bisect run <verify-cmd>` when a known-good commit exists.
37
+ - Prefer reading real state over reasoning about imagined state. The bug exists precisely because the mental model and reality differ — trust output.
38
+ - Dead hypotheses stay in the ledger marked refuted, so they are not silently retried an hour later.
39
+
40
+ ## 4. One variable per iteration
41
+
42
+ - Change one thing, re-run the reproduction, record the result in the ledger.
43
+ - A fix attempt that failed gets **reverted before the next attempt** — stacked failed fixes create a second bug on top of the first.
44
+ - Two consecutive failed fixes on the same hypothesis means the hypothesis is dead, not unlucky. Widen the search: environment, dependency versions, input data, concurrency, or the test itself being wrong.
45
+
46
+ ## 5. Close out
47
+
48
+ - The reproduction becomes a committed regression test: **red before the fix, green after** — run it both ways and quote both results. This proves the fix touched the actual cause. The mechanical form lives in the done skill's `scripts/` dir — run its `scripts/revert-probe.sh "<verify-cmd>"` from wherever that skill is installed (classic: `~/.claude/skills/done/`; plugin and `npx skills` installs keep it next to the done SKILL.md).
49
+ - Remove all instrumentation (prints, sleeps, debug flags) — check the diff for it explicitly.
50
+ - Run the full verify tier, not just the one test.
51
+ - If the session surfaced something reading the code cannot teach (a footgun, an environment quirk, a disproven approach), run `/retro` so the next session does not pay for this one's dead ends — for a debugged failure specifically, `/retro` records it in `docs/lessons.md` (symptom → cause → proven-by → fix), which is exactly what step 3 reads next time.
@@ -0,0 +1,57 @@
1
+ ---
2
+ name: done
3
+ description: Closeout ritual before declaring any non-trivial task complete. Use when about to say "done", "finished", "it works now", before opening a PR, when the user asks "is it done?", "wrap it up", "ปิดงาน" — or whenever a change is about to be handed back as complete. Not for trivial single-line answers with no code change.
4
+ ---
5
+
6
+ # Done — prove it before you say it
7
+
8
+ The doctrine says: *done is proven by a command, not by my judgment.* This is the ritual that turns that rule into a checklist. Run every step; skipping one is how "done" ships broken.
9
+
10
+ ## 1. Full verify
11
+
12
+ Run the **full** tier (`verify-full` if the repo has two tiers, else the verify command). Quote the shortest decisive line of real output.
13
+
14
+ - Red → you are not here yet. Go back to the loop; do not continue this ritual.
15
+ - No verify command exists → that is the first bug (`/luciazero-bootstrap`). Say so instead of declaring done.
16
+ - The command must actually have run **now**, in this session — a green from an hour ago proves the past, not the present.
17
+
18
+ ## 2. Skeptic diff pass
19
+
20
+ Re-read the final diff as a hostile reviewer. Tests prove what they cover; hunt what they do not:
21
+
22
+ - **Edge cases** — empty, zero, negative, unicode, first/last, concurrent
23
+ - **Error paths** — the call fails, the file is missing, the network drops; are errors swallowed?
24
+ - **Changed contracts** — public API shape, serialized formats, schema, config keys: who consumes the old shape?
25
+ - **Accidental content** — files touched by mistake, debug prints, commented-out code, leftover instrumentation, loosened dependency pins, secrets
26
+ - **Test honesty** — would the new/changed tests fail if the change were reverted? The mechanical form: `scripts/revert-probe.sh "<verify-cmd>"` answers it in one command. Weakened or deleted checks are findings, not cleanup.
27
+
28
+ Fix what you find, re-run step 1, then continue.
29
+
30
+ ## 3. Independent review, if the diff earns it
31
+
32
+ Get an adversarial second opinion — the harness's built-in review command (Claude Code: `/code-review`) or the `reviewer` agent — when **any** of these hold:
33
+
34
+ - Touches a public API, data migration, auth, money, or concurrency
35
+ - Wide diff (many files, or a subsystem you did not previously know)
36
+ - You cannot explain in one sentence why each hunk is safe
37
+
38
+ Findings go back through step 1. For small well-understood diffs, step 2 suffices — do not add ceremony the diff does not need.
39
+
40
+ ## 4. Scope check
41
+
42
+ Re-read the original request. For each part: delivered, or named as left out with the reason. Silently dropped scope is the failure mode this step exists to catch. Anything left out gets said **plainly** in the report, not buried.
43
+
44
+ ## 5. Lessons
45
+
46
+ If the session hit a dead end, a footgun, or disproved a tempting approach — run `/retro` now, while the evidence is fresh. If unfinished work remains for a future session, `/handoff` instead.
47
+
48
+ ## 6. Report
49
+
50
+ ```
51
+ Done: <what changed, one line>
52
+ Proof: <verify command> → <decisive output line>
53
+ Not covered: <what verify does not prove>
54
+ Left out: <scope not delivered + why, or "nothing">
55
+ ```
56
+
57
+ No hedging in the report: if all steps passed, state it plainly; if one did not, the task is not done and the report says what remains instead.
@@ -0,0 +1,95 @@
1
+ #!/usr/bin/env bash
2
+ # revert-probe.sh — the mechanical form of the done-skill question "would the
3
+ # new tests fail if the change were reverted?" (doctrine: red before green).
4
+ # Checks <base-ref> out into a throwaway worktree, copies ONLY the test files
5
+ # changed since <base-ref> from the working tree on top of it, and runs the
6
+ # verify command there. The result is INVERTED: old code failing the new
7
+ # tests is the PASS.
8
+ #
9
+ # Usage: revert-probe.sh "<verify-cmd>" [base-ref] (base-ref default: HEAD)
10
+ # Run it BEFORE committing — the fix and its new tests sit in the working
11
+ # tree while HEAD is still the old code. For an already-committed fix, pass
12
+ # the pre-fix ref (e.g. HEAD~1) as base-ref.
13
+ #
14
+ # Exit: 0 tests bite · 1 tests stay green on old code, or no changed test
15
+ # files · 2 UNASSESSABLE (not a git repo, no commits, invalid base).
16
+ # Pure bash + git; never touches the caller's working tree.
17
+ set -euo pipefail
18
+
19
+ VERIFY="${1:?usage: revert-probe.sh \"<verify-cmd>\" [base-ref]}"
20
+ BASE="${2:-HEAD}"
21
+
22
+ unassessable() { echo "UNASSESSABLE: $*"; exit 2; }
23
+
24
+ git rev-parse --git-dir >/dev/null 2>&1 || unassessable "not a git repo"
25
+ git rev-parse --verify HEAD >/dev/null 2>&1 || unassessable "no commits yet"
26
+ git rev-parse --verify --quiet "${BASE}^{commit}" >/dev/null 2>&1 \
27
+ || unassessable "invalid base ref: ${BASE}"
28
+ TOP="$(git rev-parse --show-toplevel 2>/dev/null)" || unassessable "no working tree (bare repo?)"
29
+ cd "${TOP}"
30
+
31
+ # test-file patterns mirror luciazero-bootstrap's detect.sh: tests-style dirs
32
+ # plus test_*.*, *_test.*, *.test.*, *.spec.* file names
33
+ is_test_file() {
34
+ case "/$1" in
35
+ */tests/*|*/test/*|*/spec/*|*/__tests__/*) return 0 ;;
36
+ esac
37
+ case "${1##*/}" in
38
+ test_*.*|*_test.*|*.test.*|*.spec.*) return 0 ;;
39
+ esac
40
+ return 1
41
+ }
42
+
43
+ # scratch space first — the changed-file list is stored NUL-delimited in a
44
+ # file, because git C-quotes non-ASCII/backslash names in its plain output
45
+ # (-z emits them raw) and bash variables cannot hold NUL bytes
46
+ TMP="$(mktemp -d)"
47
+ WT="${TMP}/worktree"
48
+ trap 'git worktree remove --force "${WT}" >/dev/null 2>&1 || true
49
+ rm -rf "${TMP}"
50
+ git worktree prune >/dev/null 2>&1 || true' EXIT
51
+
52
+ # changed vs base (tracked) plus untracked — the two sets are disjoint —
53
+ # filtered to test files that still exist (a deleted test cannot bite)
54
+ TEST_LIST="${TMP}/tests"
55
+ : > "${TEST_LIST}"
56
+ COUNT=0
57
+ while IFS= read -r -d '' F; do
58
+ [ -f "${F}" ] || continue
59
+ if is_test_file "${F}"; then
60
+ printf '%s\0' "${F}" >> "${TEST_LIST}"
61
+ COUNT=$((COUNT + 1))
62
+ fi
63
+ done < <({ git diff --name-only -z "${BASE}" --; git ls-files --others --exclude-standard -z; } 2>/dev/null)
64
+
65
+ if [ "${COUNT}" -eq 0 ]; then
66
+ echo "FAIL: no test files changed since ${BASE} — the change ships without a test that bites"
67
+ exit 1
68
+ fi
69
+
70
+ # old code in a throwaway worktree; cleanup runs on every exit path
71
+ git worktree add --detach "${WT}" "${BASE}" >/dev/null 2>&1 \
72
+ || unassessable "git worktree add failed for ${BASE}"
73
+
74
+ # overlay ONLY the changed test files from the working tree
75
+ while IFS= read -r -d '' F; do
76
+ mkdir -p "${WT}/$(dirname "${F}")"
77
+ cp "${F}" "${WT}/${F}"
78
+ done < "${TEST_LIST}"
79
+
80
+ RC=0
81
+ OUT="$(cd "${WT}" && sh -c "${VERIFY}" 2>&1)" || RC=$?
82
+
83
+ if [ "${RC}" -ne 0 ]; then
84
+ LAST=""
85
+ while IFS= read -r LINE; do
86
+ case "${LINE}" in *[![:space:]]*) LAST="${LINE}" ;; esac
87
+ done <<EOF
88
+ ${OUT}
89
+ EOF
90
+ echo "PASS: regression tests bite — old code fails the new tests"
91
+ echo " evidence (exit ${RC}): ${LAST:-<no output>}"
92
+ exit 0
93
+ fi
94
+ echo "FAIL: the changed tests stay green against the old code — they do not cover the change"
95
+ exit 1
@@ -0,0 +1,44 @@
1
+ ---
2
+ name: experiment
3
+ description: Measured-change protocol for performance and tuning work. Use when the task is "make it faster", "optimize", "reduce memory", "ทดลอง", when comparing two approaches, or whenever a claim like "this should be faster" is about to be made without a number. Not for correctness bugs — that is /debug.
4
+ ---
5
+
6
+ # Experiment — no claim without a measurement
7
+
8
+ An optimization without a baseline is a guess with confidence. The protocol is the same loop as always — but *verify* here means **measure**, and the doctrine's "never re-derive a dead end twice" means null results get recorded with the same weight as wins.
9
+
10
+ ## 1. Define the metric before touching code
11
+
12
+ - One command that prints the number: runtime, RSS, p95 latency, binary size, query count. If no such command exists, building it is step zero (the measurement twin of "no verify command is the first bug").
13
+ - Decide **now** what improvement would count — "worth it if ≥10% faster" — so the verdict is not negotiated after the numbers exist.
14
+
15
+ ## 2. Baseline
16
+
17
+ - Run the measurement **at least 3 times**; record all values, not just the mean — the spread is what separates signal from noise.
18
+ - Pin what you can: fixed seed, same input data, warm/cold state chosen deliberately, machine as quiet as you can get it. Note what you could not pin.
19
+ - Correctness verify must be green before and after — a fast wrong answer is not an optimization.
20
+
21
+ ## 3. One variable per experiment
22
+
23
+ Change one thing. Two changes in one measurement produce a number that explains neither. (Same discipline as `/debug`; same reason.)
24
+
25
+ ## 4. Measure again
26
+
27
+ - Same command, same repetitions, same conditions.
28
+ - The difference counts only if it clearly beats the baseline spread. Inside the noise = **null result**, not "slightly faster".
29
+
30
+ ## 5. Verdict and record
31
+
32
+ Append to `docs/experiments.md` (create it if absent; follow the repo's existing log if one exists):
33
+
34
+ ```
35
+ ## <date> — <hypothesis, one line>
36
+ change: <what was changed, file/approach>
37
+ baseline: <values> | result: <values>
38
+ verdict: WIN <n%> | NULL (inside noise) | LOSS
39
+ decision: <kept / reverted> — <one-line reason>
40
+ ```
41
+
42
+ - **Losers and nulls are reverted immediately** — the log keeps the knowledge, the tree keeps only wins.
43
+ - A null result is a finding: "tried X, no measurable gain — do not retry without new evidence" saves the next session the same hour. If it is load-bearing, surface it via `/retro` into the project notes too.
44
+ - Never delete a previous entry; if a new experiment overturns an old one, add the new entry and cross-reference.
@@ -0,0 +1,56 @@
1
+ ---
2
+ name: handoff
3
+ description: Write a state capsule so the next session — or a different agent/harness — can resume unfinished work without re-deriving context. Use when a session is ending with work incomplete, when the user says "handoff", "pack up", "ส่งต่อ", "continue tomorrow", when switching between Claude Code and Codex mid-task, or when context is about to be compacted away on a long task.
4
+ ---
5
+
6
+ # Handoff — state that survives the session
7
+
8
+ `/retro` records **permanent lessons**; this records **transient state**. A good capsule lets a stranger (including future-you with zero context) type one command and be productive in two minutes. A stale capsule is worse than none — so capsules are consumed and deleted, never accumulated.
9
+
10
+ ## 1. Write the capsule
11
+
12
+ Create `HANDOFF.md` at the repo root:
13
+
14
+ ```markdown
15
+ # Handoff — <date>
16
+
17
+ ## Goal
18
+ <the original request, one paragraph, verbatim enough to re-anchor>
19
+
20
+ ## State
21
+ - Done: <what is finished, and the verify evidence: command → decisive line>
22
+ - In progress: <the exact piece mid-flight, and which files hold it>
23
+ - Verify: <the command(s) to run, fast and full tier>
24
+
25
+ ## Next step
26
+ <ONE literal command or edit to do first — not a theme, an action>
27
+
28
+ ## Open hypotheses
29
+ - H1: <suspected cause / approach> — status: <untested | supported by X | refuted by Y>
30
+
31
+ ## Landmines
32
+ - <thing that looks safe but is not, discovered this session>
33
+ ```
34
+
35
+ Rules:
36
+
37
+ - **The next step is literal.** "Continue the refactor" is not a next step; "run `pytest tests/test_auth.py -k refresh` — it is the failing one" is.
38
+ - **State only what this session knows.** No aspirations, no backlog — that belongs in the issue tracker.
39
+ - **Refuted hypotheses stay in the capsule** — they are exactly what the next session would otherwise waste an hour re-deriving. If a lesson is permanent (true beyond this task), it also goes through `/retro`.
40
+ - Uncommitted changes: say so explicitly, and name the files. The capsule must not imply a clean tree that is not clean.
41
+
42
+ ## 2. Route it
43
+
44
+ - **Same machine, next session** — `HANDOFF.md` in the repo root is enough. Do not commit it.
45
+ - **Cross-machine or another person/agent** — commit it on the working branch (it travels with the code), and say in the final message that it exists.
46
+ - The project's notes file does **not** get the capsule — notes are permanent, capsules are transient. A one-line pointer is fine if the repo's convention wants one.
47
+
48
+ ## 3. Consume protocol (for the reader)
49
+
50
+ A session that finds `HANDOFF.md`:
51
+
52
+ 1. Read it **before** touching the code.
53
+ 2. Run the Verify command(s) to confirm the described state is still true — the capsule describes the past; the tree is the truth.
54
+ 3. **Delete the capsule** (or `git rm` on the branch) once absorbed. Never leave a consumed capsule to go stale; never update one incrementally across many sessions — write a fresh one at each handoff.
55
+
56
+ If the capsule and the tree disagree, trust the tree and say so.