claude-dev-env 8.36.3 → 8.37.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/codex-compatibility.md +1 -1
- package/docs/rule-guides/bdd.md +28 -0
- package/docs/rule-guides/code-standards.md +38 -0
- package/docs/rule-guides/correction-lens-excerpt.md +35 -0
- package/docs/rule-guides/doc-inventory-integrity.md +39 -0
- package/docs/rule-guides/docstring-prose-matches-implementation.md +49 -0
- package/docs/rule-guides/failure-blast-radius.md +123 -0
- package/docs/rule-guides/orphan-css-class.md +25 -0
- package/docs/rule-guides/paired-test-coverage.md +39 -0
- package/docs/rule-guides/plain-illustrative-docstrings.md +88 -0
- package/docs/rule-guides/windows-filesystem-safe.md +11 -0
- package/hooks/hooks.json +10 -0
- package/hooks/hooks_constants/spawn_readiness_hook_constants.py +111 -0
- package/hooks/routing/spawn_readiness_hook.py +195 -0
- package/hooks/routing/spawn_readiness_steps.py +266 -0
- package/hooks/routing/test_spawn_readiness_hook.py +278 -0
- package/hooks/routing/test_spawn_readiness_steps.py +87 -0
- package/package.json +1 -1
- package/rules/bdd.md +2 -23
- package/rules/code-standards.md +3 -32
- package/rules/correction-lens.md +2 -30
- package/rules/doc-inventory-integrity.md +6 -32
- package/rules/docstring-prose-matches-implementation.md +4 -42
- package/rules/failure-blast-radius.md +4 -114
- package/rules/orphan-css-class.md +4 -18
- package/rules/paired-test-coverage.md +4 -32
- package/rules/plain-illustrative-docstrings.md +4 -81
- package/rules/windows-filesystem-safe.md +3 -5
- package/scripts/codex_compat_materializer.py +4 -1
- package/scripts/tests/test_codex_compat_materializer.py +18 -18
|
@@ -0,0 +1,278 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import os
|
|
3
|
+
import subprocess
|
|
4
|
+
import sys
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
import spawn_readiness_hook
|
|
10
|
+
from hooks_constants.spawn_readiness_hook_constants import (
|
|
11
|
+
MISSING_INTERVIEW_REASON,
|
|
12
|
+
MISSING_INVESTIGATION_REASON,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
HOOK_SCRIPT = Path(__file__).resolve().parent / "spawn_readiness_hook.py"
|
|
16
|
+
THREAD_SPAWN_TOOL_NAME = "mcp__hearthbot__start_thread_session"
|
|
17
|
+
SETTLED_LINE = "Scope settled: the request names the file, the fix, and the test."
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _user(text: str) -> dict[str, object]:
|
|
21
|
+
return {"type": "user", "message": {"role": "user", "content": text}}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _tool_use(
|
|
25
|
+
name: str, tool_input: dict[str, object], block_id: str = "toolu_x"
|
|
26
|
+
) -> dict[str, object]:
|
|
27
|
+
return {
|
|
28
|
+
"type": "assistant",
|
|
29
|
+
"message": {
|
|
30
|
+
"role": "assistant",
|
|
31
|
+
"content": [{"type": "tool_use", "id": block_id, "name": name, "input": tool_input}],
|
|
32
|
+
},
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _tool_result(tool_use_id: str, is_error: bool = False) -> dict[str, object]:
|
|
37
|
+
return {
|
|
38
|
+
"type": "user",
|
|
39
|
+
"message": {
|
|
40
|
+
"role": "user",
|
|
41
|
+
"content": [
|
|
42
|
+
{
|
|
43
|
+
"type": "tool_result",
|
|
44
|
+
"tool_use_id": tool_use_id,
|
|
45
|
+
"content": "ok",
|
|
46
|
+
"is_error": is_error,
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
},
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _queued(prompt: str) -> dict[str, object]:
|
|
54
|
+
return {"type": "attachment", "attachment": {"type": "queued_command", "prompt": prompt}}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
READ_STEP = _tool_use("Bash", {"command": "cat README.md"})
|
|
58
|
+
QUESTION_STEP = _tool_use("mcp__hearthbot__ask_decision", {"question": "Which repo should this land in?"})
|
|
59
|
+
REPLY_STEP = _user("The public one.")
|
|
60
|
+
INTERVIEWED_TRANSCRIPT = [_user("Build the hook."), READ_STEP, QUESTION_STEP, REPLY_STEP]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _spawn_input(tool_name: str, brief: str = "Do the task.") -> dict[str, object]:
|
|
64
|
+
if tool_name == THREAD_SPAWN_TOOL_NAME:
|
|
65
|
+
return {"title": "Task", "instructions": brief}
|
|
66
|
+
return {"description": "Task", "prompt": brief}
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _write_transcript(tmp_path: Path, all_entries: list[dict[str, object]]) -> Path:
|
|
70
|
+
transcript_path = tmp_path / "transcript.jsonl"
|
|
71
|
+
transcript_path.write_text(
|
|
72
|
+
"".join(json.dumps(each_entry) + "\n" for each_entry in all_entries), encoding="utf-8"
|
|
73
|
+
)
|
|
74
|
+
return transcript_path
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _run_hook(tmp_path: Path, hook_payload: dict[str, object]) -> str:
|
|
78
|
+
completed = subprocess.run(
|
|
79
|
+
[sys.executable, str(HOOK_SCRIPT)],
|
|
80
|
+
input=json.dumps(
|
|
81
|
+
{"hook_event_name": "PreToolUse", "tool_use_id": "toolu_spawn", **hook_payload}
|
|
82
|
+
),
|
|
83
|
+
capture_output=True,
|
|
84
|
+
text=True,
|
|
85
|
+
env={**os.environ, "HOME": str(tmp_path)},
|
|
86
|
+
check=True,
|
|
87
|
+
)
|
|
88
|
+
return completed.stdout
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _spawn(
|
|
92
|
+
tmp_path: Path,
|
|
93
|
+
all_entries: list[dict[str, object]],
|
|
94
|
+
tool_name: str = "Agent",
|
|
95
|
+
brief: str = "Do the task.",
|
|
96
|
+
) -> str:
|
|
97
|
+
return _run_hook(
|
|
98
|
+
tmp_path,
|
|
99
|
+
{
|
|
100
|
+
"tool_name": tool_name,
|
|
101
|
+
"tool_input": _spawn_input(tool_name, brief),
|
|
102
|
+
"transcript_path": str(_write_transcript(tmp_path, all_entries)),
|
|
103
|
+
},
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _deny_reason(stdout: str) -> str:
|
|
108
|
+
decision = json.loads(stdout)["hookSpecificOutput"]
|
|
109
|
+
assert decision["hookEventName"] == "PreToolUse"
|
|
110
|
+
assert decision["permissionDecision"] == "deny"
|
|
111
|
+
return decision["permissionDecisionReason"]
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _decision_log(tmp_path: Path) -> list[dict[str, object]]:
|
|
115
|
+
log_path = tmp_path / ".claude" / "logs" / "spawn-readiness.jsonl"
|
|
116
|
+
return [
|
|
117
|
+
json.loads(each_line) for each_line in log_path.read_text(encoding="utf-8").splitlines()
|
|
118
|
+
]
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
@pytest.mark.parametrize("tool_name", ["Agent", "Task", THREAD_SPAWN_TOOL_NAME])
|
|
122
|
+
def test_should_let_the_spawn_run_after_a_read_a_question_and_a_reply(
|
|
123
|
+
tmp_path: Path, tool_name: str
|
|
124
|
+
) -> None:
|
|
125
|
+
assert _spawn(tmp_path, INTERVIEWED_TRANSCRIPT, tool_name) == ""
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def test_should_deny_a_spawn_with_no_read_since_the_request(tmp_path: Path) -> None:
|
|
129
|
+
transcript = [_user("Build the hook."), QUESTION_STEP, REPLY_STEP]
|
|
130
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INVESTIGATION_REASON
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def test_should_deny_a_spawn_with_no_question_to_the_user(tmp_path: Path) -> None:
|
|
134
|
+
transcript = [_user("Build the hook."), READ_STEP]
|
|
135
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def test_should_deny_a_spawn_whose_question_has_no_reply_yet(tmp_path: Path) -> None:
|
|
139
|
+
transcript = [_user("Build the hook."), READ_STEP, QUESTION_STEP]
|
|
140
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def test_should_name_both_gaps_when_the_session_did_neither(tmp_path: Path) -> None:
|
|
144
|
+
reason = _deny_reason(_spawn(tmp_path, [_user("Build the hook.")]))
|
|
145
|
+
assert reason == f"{MISSING_INVESTIGATION_REASON} {MISSING_INTERVIEW_REASON}"
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def test_should_count_a_plain_chat_question_as_no_interview(
|
|
149
|
+
tmp_path: Path,
|
|
150
|
+
) -> None:
|
|
151
|
+
statement = _tool_use("mcp__hearthbot__post_message", {"text": "Which repo should this land in?"})
|
|
152
|
+
transcript = [_user("Build the hook."), READ_STEP, statement, REPLY_STEP, READ_STEP]
|
|
153
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def test_should_count_an_answered_ask_user_question_as_the_interview(tmp_path: Path) -> None:
|
|
157
|
+
transcript = [
|
|
158
|
+
_user("Build the hook."),
|
|
159
|
+
_tool_use("Read", {"file_path": "README.md"}),
|
|
160
|
+
_tool_use("AskUserQuestion", {"questions": []}, block_id="toolu_ask"),
|
|
161
|
+
_tool_result("toolu_ask"),
|
|
162
|
+
]
|
|
163
|
+
assert _spawn(tmp_path, transcript) == ""
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def test_should_count_an_answered_widget_as_the_interview(tmp_path: Path) -> None:
|
|
167
|
+
widget = _tool_use("mcp__hearthbot__post_widget", {"family": "visualize", "input": {}})
|
|
168
|
+
transcript = [_user("Build the hook."), READ_STEP, widget, _user("The left layout.")]
|
|
169
|
+
assert _spawn(tmp_path, transcript) == ""
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def test_should_count_an_answered_decision_card_as_the_interview(tmp_path: Path) -> None:
|
|
173
|
+
transcript = [
|
|
174
|
+
_user("Build the hook."),
|
|
175
|
+
_tool_use("mcp__hearthbot__fetch_thread", {"thread_id": "t"}),
|
|
176
|
+
_tool_use("mcp__hearthbot__ask_decision", {"question": "Pick one"}),
|
|
177
|
+
_queued('<wake reason="mention"><message from="human">Option one.</message></wake>'),
|
|
178
|
+
]
|
|
179
|
+
assert _spawn(tmp_path, transcript, THREAD_SPAWN_TOOL_NAME) == ""
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def test_should_skip_an_agent_relay_as_the_reply(tmp_path: Path) -> None:
|
|
183
|
+
relay = _queued('<relay from="coordinator"><note>Keep going.</note></relay>')
|
|
184
|
+
transcript = [_user("Build the hook."), READ_STEP, QUESTION_STEP, relay]
|
|
185
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def test_should_require_a_new_interview_for_a_follow_up_request(tmp_path: Path) -> None:
|
|
189
|
+
earlier_spawn = _tool_use("Agent", {"prompt": "Build the hook."})
|
|
190
|
+
transcript = [*INTERVIEWED_TRANSCRIPT, earlier_spawn, _user("Now add a second hook."), READ_STEP]
|
|
191
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def test_should_treat_two_messages_in_a_row_after_a_question_as_one_reply(
|
|
195
|
+
tmp_path: Path,
|
|
196
|
+
) -> None:
|
|
197
|
+
transcript = [*INTERVIEWED_TRANSCRIPT, _user("Also keep it short.")]
|
|
198
|
+
assert _spawn(tmp_path, transcript) == ""
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def test_should_pass_the_interview_on_a_scope_settled_line_and_log_it(tmp_path: Path) -> None:
|
|
202
|
+
transcript = [_user("Fix the typo in README.md line 3."), READ_STEP]
|
|
203
|
+
brief = f"Fix the typo.\n{SETTLED_LINE}"
|
|
204
|
+
assert _spawn(tmp_path, transcript, THREAD_SPAWN_TOOL_NAME, brief) == ""
|
|
205
|
+
[log_record] = _decision_log(tmp_path)
|
|
206
|
+
assert log_record["outcome"] == "scope_settled"
|
|
207
|
+
assert log_record["tool_name"] == THREAD_SPAWN_TOOL_NAME
|
|
208
|
+
assert log_record["scope_settled_line"] == SETTLED_LINE
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def test_should_still_require_the_read_with_a_scope_settled_line(tmp_path: Path) -> None:
|
|
212
|
+
brief = f"Fix the typo.\n{SETTLED_LINE}"
|
|
213
|
+
reason = _deny_reason(_spawn(tmp_path, [_user("Fix the typo.")], brief=brief))
|
|
214
|
+
assert reason == MISSING_INVESTIGATION_REASON
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def test_should_deny_a_scope_settled_line_with_no_reason(tmp_path: Path) -> None:
|
|
218
|
+
transcript = [_user("Fix the typo."), READ_STEP]
|
|
219
|
+
reason = _deny_reason(_spawn(tmp_path, transcript, brief="Fix it.\nScope settled:"))
|
|
220
|
+
assert reason == MISSING_INTERVIEW_REASON
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def test_should_let_a_read_only_subagent_run_without_checks(tmp_path: Path) -> None:
|
|
224
|
+
stdout = _run_hook(
|
|
225
|
+
tmp_path,
|
|
226
|
+
{
|
|
227
|
+
"tool_name": "Agent",
|
|
228
|
+
"tool_input": {"subagent_type": "Explore", "prompt": "Find the hooks."},
|
|
229
|
+
"transcript_path": str(_write_transcript(tmp_path, [_user("Build the hook.")])),
|
|
230
|
+
},
|
|
231
|
+
)
|
|
232
|
+
assert stdout == ""
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def test_should_count_a_read_only_subagent_as_the_investigation(tmp_path: Path) -> None:
|
|
236
|
+
explore = _tool_use("Agent", {"subagent_type": "Explore", "prompt": "Find it."})
|
|
237
|
+
transcript = [_user("Build the hook."), explore, QUESTION_STEP, REPLY_STEP]
|
|
238
|
+
assert _spawn(tmp_path, transcript) == ""
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def test_should_let_a_spawn_from_inside_a_subagent_run(tmp_path: Path) -> None:
|
|
242
|
+
stdout = _run_hook(
|
|
243
|
+
tmp_path,
|
|
244
|
+
{
|
|
245
|
+
"tool_name": "Agent",
|
|
246
|
+
"agent_id": "agent_1",
|
|
247
|
+
"tool_input": _spawn_input("Agent"),
|
|
248
|
+
"transcript_path": str(_write_transcript(tmp_path, [_user("Build the hook.")])),
|
|
249
|
+
},
|
|
250
|
+
)
|
|
251
|
+
assert stdout == ""
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def test_should_let_the_spawn_run_and_log_when_the_transcript_is_unreadable(
|
|
255
|
+
tmp_path: Path,
|
|
256
|
+
) -> None:
|
|
257
|
+
stdout = _run_hook(
|
|
258
|
+
tmp_path,
|
|
259
|
+
{
|
|
260
|
+
"tool_name": "Agent",
|
|
261
|
+
"tool_input": _spawn_input("Agent"),
|
|
262
|
+
"transcript_path": str(tmp_path / "absent.jsonl"),
|
|
263
|
+
},
|
|
264
|
+
)
|
|
265
|
+
assert stdout == ""
|
|
266
|
+
[log_record] = _decision_log(tmp_path)
|
|
267
|
+
assert log_record["outcome"] == "transcript_unreadable"
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def test_should_return_the_scope_settled_line_from_the_brief() -> None:
|
|
271
|
+
tool_input = {"instructions": f"Do X.\n {SETTLED_LINE}"}
|
|
272
|
+
assert (
|
|
273
|
+
spawn_readiness_hook.scope_settled_line(THREAD_SPAWN_TOOL_NAME, tool_input) == SETTLED_LINE
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def test_should_ignore_a_tool_it_does_not_check() -> None:
|
|
278
|
+
assert spawn_readiness_hook.decide_hook_output({"tool_name": "Read", "tool_input": {}}) is None
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
from spawn_readiness_steps import SessionStep, readiness_gaps, session_steps
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def _user(text: str) -> dict[str, object]:
|
|
7
|
+
return {"type": "user", "message": {"role": "user", "content": text}}
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _tool_use(name: str, tool_input: dict[str, object], block_id: str = "toolu_x") -> dict[str, object]:
|
|
11
|
+
return {
|
|
12
|
+
"type": "assistant",
|
|
13
|
+
"message": {
|
|
14
|
+
"role": "assistant",
|
|
15
|
+
"content": [{"type": "tool_use", "id": block_id, "name": name, "input": tool_input}],
|
|
16
|
+
},
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _tool_result(tool_use_id: str) -> dict[str, object]:
|
|
21
|
+
return {
|
|
22
|
+
"type": "user",
|
|
23
|
+
"message": {
|
|
24
|
+
"role": "user",
|
|
25
|
+
"content": [{"type": "tool_result", "tool_use_id": tool_use_id, "content": "ok"}],
|
|
26
|
+
},
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
READ_STEP = _tool_use("Bash", {"command": "cat README.md"})
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_should_skip_meta_entries_and_tool_results_as_user_messages() -> None:
|
|
34
|
+
all_lines = [
|
|
35
|
+
json.dumps(_user("Build the hook.")),
|
|
36
|
+
json.dumps({**_user("Base directory for this skill"), "isMeta": True}),
|
|
37
|
+
json.dumps(READ_STEP),
|
|
38
|
+
json.dumps(_tool_result("toolu_x")),
|
|
39
|
+
]
|
|
40
|
+
assert session_steps(all_lines) == [
|
|
41
|
+
SessionStep.USER_MESSAGE,
|
|
42
|
+
SessionStep.READ,
|
|
43
|
+
]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def test_should_leave_the_spawn_under_check_out_of_the_steps() -> None:
|
|
47
|
+
explore = _tool_use(
|
|
48
|
+
"Agent", {"subagent_type": "Explore", "prompt": "Find it."}, block_id="toolu_spawn"
|
|
49
|
+
)
|
|
50
|
+
all_lines = [json.dumps(_user("Build the hook.")), json.dumps(explore)]
|
|
51
|
+
assert session_steps(all_lines, "toolu_spawn") == [
|
|
52
|
+
SessionStep.USER_MESSAGE
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def test_should_report_no_gaps_for_a_reply_after_a_question() -> None:
|
|
57
|
+
all_steps = [
|
|
58
|
+
SessionStep.USER_MESSAGE,
|
|
59
|
+
SessionStep.READ,
|
|
60
|
+
SessionStep.QUESTION,
|
|
61
|
+
SessionStep.USER_MESSAGE,
|
|
62
|
+
]
|
|
63
|
+
assert readiness_gaps(all_steps) == []
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def test_should_read_a_github_get_call_as_a_read_and_a_chat_question_as_another_call() -> None:
|
|
67
|
+
all_lines = [
|
|
68
|
+
json.dumps(_tool_use("mcp__github__get_file_contents", {"path": "a"})),
|
|
69
|
+
json.dumps(_tool_use("mcp__hearthbot__reply", {"text": "Which one?"})),
|
|
70
|
+
json.dumps(_tool_use("mcp__hearthbot__ask_decision", {"question": "Which one?"})),
|
|
71
|
+
]
|
|
72
|
+
assert session_steps(all_lines) == [SessionStep.READ, SessionStep.OTHER_TOOL_CALL, SessionStep.QUESTION]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def test_should_start_the_span_at_the_request_before_an_answered_question() -> None:
|
|
76
|
+
all_steps = [
|
|
77
|
+
SessionStep.USER_MESSAGE,
|
|
78
|
+
SessionStep.READ,
|
|
79
|
+
SessionStep.OTHER_TOOL_CALL,
|
|
80
|
+
SessionStep.USER_MESSAGE,
|
|
81
|
+
SessionStep.QUESTION,
|
|
82
|
+
SessionStep.USER_MESSAGE,
|
|
83
|
+
]
|
|
84
|
+
assert readiness_gaps(all_steps) == [
|
|
85
|
+
"Investigate the request before this spawn. Read the files, threads, or sources it names, "
|
|
86
|
+
"so the brief and the agent count fit the task. Run the reads in a message before the spawn."
|
|
87
|
+
]
|
package/package.json
CHANGED
package/rules/bdd.md
CHANGED
|
@@ -9,27 +9,6 @@ paths:
|
|
|
9
9
|
|
|
10
10
|
# BDD (discovery-driven development)
|
|
11
11
|
|
|
12
|
-
|
|
12
|
+
For every non-trivial feature, discover what you do not know, illustrate it with concrete examples in chat, formulate each behavior as a narrow "should ..." specification, then build each one through the TDD loop (CODE_RULES §8). Name developer-facing specs and tests as should sentences.
|
|
13
13
|
|
|
14
|
-
**
|
|
15
|
-
|
|
16
|
-
- `@~/.claude/docs/BDD_SCENARIO_QUALITY.md` — seven scenario quality patterns (§7.6-style)
|
|
17
|
-
- `@~/.claude/docs/BDD_DISCOVERY_PROTOCOL.md` — Example Mapping algorithm for chat
|
|
18
|
-
- `@~/.claude/docs/BDD_TEST_LAYOUT.md` — describe/when/should layout and soap-opera personas
|
|
19
|
-
|
|
20
|
-
## What you do for every non-trivial feature
|
|
21
|
-
|
|
22
|
-
1. **Deliberate Discovery** — Reduce uncertainty before code; surface what you do not know (Smart & Molak §5.4).
|
|
23
|
-
2. **Illustrate** — Explore goals, constraints, and concrete examples in chat; "given … when … then …" style outcomes.
|
|
24
|
-
3. **Formulate** — Express behavior as narrow **"should …"** specifications the user can approve.
|
|
25
|
-
4. **Automate** — Build each formulated behavior with tests. Red-green-refactor is the default loop, and the TDD skill (`pstack:tdd`) carries it: CODE_RULES §8, as stated in [`code-standards.md`](code-standards.md). A prototype may run ahead of its tests and adds them before the pull request goes ready.
|
|
26
|
-
|
|
27
|
-
Conversation is the essential practice: if discovery is skipped, structured formats do not rescue the workflow (Minimal BDD).
|
|
28
|
-
|
|
29
|
-
## Solo developer
|
|
30
|
-
|
|
31
|
-
You are often the stakeholder. Use **Example Mapping** in chat ("the one where …", probes, parking lot). See the optional long-form references above for the full algorithm and anti-pattern list.
|
|
32
|
-
|
|
33
|
-
## Naming
|
|
34
|
-
|
|
35
|
-
Developer-facing specs and tests use **should** sentences so intent stays visible (Dan North, "Introducing BDD", 2006).
|
|
14
|
+
**Full text:** [`docs/rule-guides/bdd.md`](../docs/rule-guides/bdd.md), with the long-form BDD references.
|
package/rules/code-standards.md
CHANGED
|
@@ -10,37 +10,8 @@ paths:
|
|
|
10
10
|
|
|
11
11
|
# Code Standards
|
|
12
12
|
|
|
13
|
-
|
|
14
|
-
> **Checked-in pointer:** [`.cursor/BUGBOT.md`](../../../.cursor/BUGBOT.md) — the file Cursor BugBot reads; it points at `CODE_RULES.md`.
|
|
15
|
-
> **Production enforcement.** The staged policy lint runs `hooks/blocking/code_rules_enforcer.py` over each changed file. No write-time hook runs it. CI runs that lint against the merge base. Each mechanical rule carries a synchronization test.
|
|
13
|
+
[`CODE_RULES.md`](../docs/CODE_RULES.md) is the review contract for code quality. Load it when you review a pull request, resolve a policy conflict, or generate code. Red, green, refactor is the default loop (CODE_RULES §8). Keep engineering right-sized (CODE_RULES §7).
|
|
16
14
|
|
|
17
|
-
|
|
15
|
+
**Enforcement:** the staged policy lint runs `hooks/blocking/code_rules_enforcer.py` over each changed file, and CI runs it against the merge base.
|
|
18
16
|
|
|
19
|
-
|
|
20
|
-
|---|---|---|
|
|
21
|
-
| Contract | `docs/CODE_RULES.md` | Full review criteria for PR agents, loaded on demand |
|
|
22
|
-
| Pointer | `.cursor/BUGBOT.md` | Checked-in file Cursor BugBot reads; points at `CODE_RULES.md` |
|
|
23
|
-
| Enforcer | `hooks/blocking/code_rules_enforcer.py` | Hand-maintained checks the staged policy lint runs; not generated from the docs |
|
|
24
|
-
| Lint | `scripts/cde_lint.py` | Runs the enforcer and the other policy rules over staged or changed files, grading each against the file's prior text; see [`ci-owns-the-gate.md`](ci-owns-the-gate.md) for what each selection flag reports |
|
|
25
|
-
| Session rules | `rules/*.md` | Runtime session policy (questions, tasks, shell) |
|
|
26
|
-
|
|
27
|
-
Load `CODE_RULES.md` when reviewing a PR, resolving a policy conflict, or generating code. Prefer linking this ref over restating rules.
|
|
28
|
-
|
|
29
|
-
Two standards live in `CODE_RULES.md` in full:
|
|
30
|
-
|
|
31
|
-
- **TDD** — CODE_RULES §8: red, green, refactor is the default loop for a bug fix and for new behavior, and the TDD skill (`pstack:tdd`) carries the procedure. A prototype may run ahead of its tests and adds them before the pull request goes ready. A bug fix ships with a test that reproduces the bug.
|
|
32
|
-
- **Right-sized engineering** — CODE_RULES §7 / AGENTS Design: functions over classes; concrete over abstract; add an abstraction at the commit that introduces its second concrete implementation. That count is a house call, one occurrence earlier than the rule of three Fowler credits to Don Roberts. The direction comes from the literature; the number does not, so read it as this package's setting rather than as a cited standard.
|
|
33
|
-
|
|
34
|
-
BDD is the outer process and TDD is the inner loop: [`bdd.md`](bdd.md) discovers and formulates the behavior a feature needs, then each formulated behavior is built through the TDD cycle.
|
|
35
|
-
|
|
36
|
-
## Session policies (ref docs, not restated here)
|
|
37
|
-
|
|
38
|
-
| Concern | Rule file |
|
|
39
|
-
|---|---|
|
|
40
|
-
| Handling a correction from the user | [`correction-lens.md`](correction-lens.md) |
|
|
41
|
-
| Task tracking / worker completion | [`workers-done-before-complete.md`](workers-done-before-complete.md) |
|
|
42
|
-
| Multi-step task list | skill `task-build` (see agents catalog) |
|
|
43
|
-
|
|
44
|
-
## Validation
|
|
45
|
-
|
|
46
|
-
Mechanical enforcer coverage is checked by the existing `hooks/blocking/test_code_rules_enforcer*.py` suite.
|
|
17
|
+
**Full text:** [`docs/rule-guides/code-standards.md`](../docs/rule-guides/code-standards.md), with the policy surface map.
|
package/rules/correction-lens.md
CHANGED
|
@@ -96,37 +96,9 @@ exemption from the archiving procedure's side.
|
|
|
96
96
|
|
|
97
97
|
An edit that sharpens this rule is welcome. The file stays.
|
|
98
98
|
|
|
99
|
-
##
|
|
99
|
+
## Codex copy
|
|
100
100
|
|
|
101
|
-
Codex reads its repository `AGENTS.md
|
|
102
|
-
contract.
|
|
103
|
-
|
|
104
|
-
```
|
|
105
|
-
Correction handling for this run, from rules/correction-lens.md.
|
|
106
|
-
|
|
107
|
-
Every correction the user gives becomes a control. Run it through five layers
|
|
108
|
-
and land it at the highest one that can hold it:
|
|
109
|
-
|
|
110
|
-
1. Codebase. The mistake is impossible by how the code is written: a type, a
|
|
111
|
-
signature, a data structure, an API shape.
|
|
112
|
-
2. Static analysis. A program reads the tree and decides: a lint rule, an
|
|
113
|
-
enforcer check, a paired test, a CI gate.
|
|
114
|
-
3. Review tooling. A reviewer or a review bot reads the criterion: a code-rules
|
|
115
|
-
row, a BUGBOT pointer, a Graphite rule.
|
|
116
|
-
4. Skill. An agent follows a procedure.
|
|
117
|
-
5. Style guide. Word choice and prose shape.
|
|
118
|
-
|
|
119
|
-
Close the correction with three sentences: the layer chosen, why each higher
|
|
120
|
-
layer cannot hold the lesson, and the change opened at that layer in this run.
|
|
121
|
-
Effort rules out no layer; a missing capability does, and you name which one.
|
|
122
|
-
|
|
123
|
-
The same correction arriving twice means the layer was too low. Move it up one
|
|
124
|
-
layer and say so.
|
|
125
|
-
|
|
126
|
-
Land the control in the repository whose code, CI, or pipeline it guards. The
|
|
127
|
-
shared environment package takes only repo-agnostic controls that any
|
|
128
|
-
repository could use. A memory file records the decision and encodes nothing.
|
|
129
|
-
```
|
|
101
|
+
Codex reads its repository `AGENTS.md`. The standalone excerpt it receives lives in [`docs/rule-guides/correction-lens-excerpt.md`](../docs/rule-guides/correction-lens-excerpt.md).
|
|
130
102
|
|
|
131
103
|
## Sibling rules
|
|
132
104
|
|
|
@@ -11,38 +11,12 @@ paths:
|
|
|
11
11
|
|
|
12
12
|
# Documentation Inventory Integrity
|
|
13
13
|
|
|
14
|
-
A doc that inventories code
|
|
14
|
+
A doc that inventories code stays in step with the code, in the same change:
|
|
15
15
|
|
|
16
|
-
|
|
16
|
+
1. Every bare filename a per-directory `CLAUDE.md` names in a table cell or a fenced run command exists in its subtree.
|
|
17
|
+
2. A package `README.md` Layout table, `CLAUDE.md` "Key files" list, or skill `SKILL.md` Layout table names each new production file and says what it does. Broaden the purpose sentence and the file's description when its responsibility grows.
|
|
18
|
+
3. An env-var table row names a code file that reads the variable.
|
|
17
19
|
|
|
18
|
-
|
|
20
|
+
**Enforcement:** `repository_checks/claude_md.py`, `repository_checks/package_inventory.py`, and `repository_checks/env_var_documentation.py`. Run `python packages/claude-dev-env/scripts/repository_policy.py` before you commit. CI runs the same command.
|
|
19
21
|
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
It collects two kinds of reference:
|
|
23
|
-
|
|
24
|
-
- **Table cells** — the first column of each markdown table row **outside** a fenced code block, keeping cells that name a bare filename in backticks, with no path separator, not a slash-command, ending in a known extension (`.py`, `.md`, `.json`, `.mjs`, `.js`, `.ts`, `.ps1`, `.cmd`, `.ahk`, `.yml`, `.yaml`, `.sh`, `.txt`, `.cfg`, `.toml`, `.ini`).
|
|
25
|
-
- **Run commands** — each line **inside** a fenced code block that invokes an interpreter (`python`, `python.exe`, `python3`, `node`, `pwsh`, `powershell`, `bash`, `sh`, `ruby`, `perl`) on a script, taking that script's basename when it ends in `.py`, `.mjs`, `.js`, `.ts`, `.ps1`, `.sh`, `.rb`, or `.pl`.
|
|
26
|
-
|
|
27
|
-
A fenced *table row* is an example and contributes nothing; a fenced *run command* is a contract the reader runs and is checked. The write is blocked when a collected filename exists nowhere under the scan root — the `CLAUDE.md` directory's parent, covering the directory, its subdirectories, and its siblings. A filesystem error that halts the subtree walk fails open.
|
|
28
|
-
|
|
29
|
-
The check stays quiet for a target that is not a `CLAUDE.md`, for a cell holding a path, a subdirectory ending in `/`, or a slash-command, for a table row inside a fence, for an inline `python x.py` mention outside a fence, and for a table naming an explicit relative-path source (a `../` token), which documents files outside the subtree by design.
|
|
30
|
-
|
|
31
|
-
## 2. A package inventory names each new production file
|
|
32
|
-
|
|
33
|
-
A package directory that documents its own files in a `README.md` Layout table, a `CLAUDE.md` "Key files" list, or a skill `SKILL.md` Layout table keeps that inventory in step with the directory. A new production file in such a directory gets its entry — a table row or a list bullet naming the file in backticks and saying what it does — in the same change.
|
|
34
|
-
|
|
35
|
-
`repository_checks/package_inventory.py` scans the committed tree and reports a production file whose basename appears in no present inventory, loading its detection logic from `package_inventory_stale_blocker.py`. It names the fix. A skill `SKILL.md` Layout table that maps `scripts/` counts as the inventory for files in that subdirectory.
|
|
36
|
-
|
|
37
|
-
Two free-prose slices stay with judgment and belong in the same change:
|
|
38
|
-
|
|
39
|
-
1. **Purpose / scope sentence.** When the new module adds a responsibility the package `## Purpose` (or the parent inventory's one-line summary of the subdirectory) omits, broaden that sentence to name it. A hook cannot derive a module's responsibility from its filename.
|
|
40
|
-
2. **Per-file description clause.** When a file gains a responsibility the inventory's em-dash description omits — a new public function, a new module-level constant — broaden the clause to name it. The gate checks only that the basename appears once and never reads the description. Constants modules (`*_constants.py`, or any `.py` directly inside `config/`) are the common shape: the clause that lands in the module docstring lands in the inventory description in the same change. The gate fires on Write of a new file and skips files directly inside `config/`, so an Edit adding a constant to an existing config module matches neither path.
|
|
41
|
-
|
|
42
|
-
This is the `category-o-docstring-vs-impl-drift` (O8) orphaned-doc-claim shape applied to a package inventory.
|
|
43
|
-
|
|
44
|
-
## 3. An env-var table row names a file that reads the variable
|
|
45
|
-
|
|
46
|
-
Every row in an env-var summary table pairs an UPPER_SNAKE variable with a code-file path that reads it — written as `` | `GOOGLE_APPLICATION_CREDENTIALS` | `auth/google_auth.py` | … | ``. When a code change removes the last read of a variable from a file, the same change drops or corrects the row naming that file.
|
|
47
|
-
|
|
48
|
-
`repository_checks/env_var_documentation.py` scans tracked `.md` files and reports a row whose named code file exists yet never references the variable, loading its detection logic from `env_var_table_code_drift_blocker.py`. It names the fix. A row whose code file resolves nowhere stays quiet, since the check cannot prove the drift.
|
|
22
|
+
**Full text:** [`docs/rule-guides/doc-inventory-integrity.md`](../docs/rule-guides/doc-inventory-integrity.md).
|
|
@@ -6,48 +6,10 @@ paths:
|
|
|
6
6
|
|
|
7
7
|
# Docstring Prose Matches Implementation
|
|
8
8
|
|
|
9
|
-
**When this applies:**
|
|
9
|
+
**When this applies:** Writing or editing a docstring, or a skill's companion `SKILL.md`, whose prose lists the inputs, matches, skips, or step order a body applies.
|
|
10
10
|
|
|
11
|
-
|
|
11
|
+
The list covers every behavior the body applies, and the body accepts only what the list names. When the body changes the set, update the prose in the same edit. A hook's docstring and `CORRECTIVE_MESSAGE` claim exactly the shapes its detector flags.
|
|
12
12
|
|
|
13
|
-
|
|
13
|
+
**Enforcement:** `code_rules_docstrings.py` and the JS slices in `code_rules_imports_logging.py`, which the staged policy lint runs through `code_rules_enforcer.py`, plus its `hook-prose-consistency` rule. CI runs it against the merge base.
|
|
14
14
|
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
## Write-time checks
|
|
18
|
-
|
|
19
|
-
Read the body and the docstring side by side. Apply each check that matches the prose:
|
|
20
|
-
|
|
21
|
-
- **Unions / match sources** — every member of a "what counts" union appears in the prose.
|
|
22
|
-
- **Suppressors / skip lists** — every early-return suppressor appears in the prose.
|
|
23
|
-
- **Step order** — named order matches call order; branch-guarded corrective steps are named too.
|
|
24
|
-
- **Shared fallbacks** — every condition that reaches a fallback call is named.
|
|
25
|
-
- **Predicate breadth** — the body accepts only the inputs the prose names.
|
|
26
|
-
- **Exclusion axis** — an exclusion clause keys on the same axis the body classifies on.
|
|
27
|
-
- **Companion docs** — a `SKILL.md` (or sibling) order/content claim matches the producer body.
|
|
28
|
-
- **Gate-outcome status flags** — an outcome routed to a blocker (`blocker = ...; break`) reads as blocked everywhere, never as a bypass.
|
|
29
|
-
- **Returns / Raises / Note claims** — each free-form claim matches the body.
|
|
30
|
-
|
|
31
|
-
Many deterministic shapes of this drift are checked in `code_rules_docstrings.py` (and the JS and `.mjs` slices in `code_rules_imports_logging.py`). The staged policy lint reaches both through `code_rules_enforcer.py`, and CI runs it against the merge base. Free-form rest is judgment.
|
|
32
|
-
|
|
33
|
-
## Hook prose matches its detector
|
|
34
|
-
|
|
35
|
-
A hook module is the sharpest case of the same rule: its docstring lead narrative and its `CORRECTIVE_MESSAGE` describe exactly the shapes the detector flags, and claim no broader trigger surface than the regex enforces.
|
|
36
|
-
|
|
37
|
-
The staged policy lint carries this as its `hook-prose-consistency` rule, covering hook modules and their `*_constants.py` companions. It reports prose that claims a trigger the detector never fires on, and names the fix. No write-time hook runs it, so CI is where it reports.
|
|
38
|
-
|
|
39
|
-
After writing a hook, ask: would a token matching every word of this message trip the detector? When the message names a shape the regex skips, rewrite the message to name only what the regex catches. The path-shape case is the common overstatement — a detector that keys off a path separator must not claim it blocks an "output-key segment". The corrective message spells the rewrite.
|
|
40
|
-
|
|
41
|
-
## Full standard
|
|
42
|
-
|
|
43
|
-
The full Category O judgment standard — sub-buckets O1–O9, the complete write-time gate inventory, free-form checklists, and worked examples — lives in:
|
|
44
|
-
|
|
45
|
-
`~/.claude/audit-rubrics/category_rubrics/category-o-docstring-vs-impl-drift.md`
|
|
46
|
-
|
|
47
|
-
## Division of labor
|
|
48
|
-
|
|
49
|
-
| Surface | Role |
|
|
50
|
-
|---|---|
|
|
51
|
-
| **This rule** | Always-on write-time policy and the compact checklist above. |
|
|
52
|
-
| Category O rubric | Single thick source for the full standard (on demand). |
|
|
53
|
-
| Category O prompt | Audit template; points at the rubric for judgment. |
|
|
15
|
+
**Full text:** [`docs/rule-guides/docstring-prose-matches-implementation.md`](../docs/rule-guides/docstring-prose-matches-implementation.md), with the write-time checklist. The Category O audit rubric carries the full standard.
|