claude-dev-env 8.36.4 → 8.37.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/rule-guides/asd-ste100-language.md +42 -0
- package/docs/rule-guides/cleanup-temp-files.md +35 -0
- package/docs/rule-guides/correction-lens.md +112 -0
- package/docs/rule-guides/destructive-commands.md +49 -0
- package/docs/rule-guides/explore-thoroughly.md +27 -0
- package/docs/rule-guides/filesystem-search.md +53 -0
- package/docs/rule-guides/memory-stores-durable-facts.md +59 -0
- package/docs/rule-guides/no-contrast-framing.md +70 -0
- package/docs/rule-guides/research-mode.md +31 -0
- package/docs/rule-guides/verify-before-asking.md +54 -0
- package/docs/rule-guides/verify-runtime-state.md +44 -0
- package/hooks/hooks.json +10 -0
- package/hooks/hooks_constants/spawn_readiness_hook_constants.py +111 -0
- package/hooks/routing/spawn_readiness_hook.py +195 -0
- package/hooks/routing/spawn_readiness_steps.py +266 -0
- package/hooks/routing/test_spawn_readiness_hook.py +278 -0
- package/hooks/routing/test_spawn_readiness_steps.py +87 -0
- package/package.json +1 -1
- package/rules/asd-ste100-language.md +5 -36
- package/rules/cleanup-temp-files.md +5 -29
- package/rules/correction-lens.md +13 -102
- package/rules/destructive-commands.md +5 -41
- package/rules/explore-thoroughly.md +5 -21
- package/rules/filesystem-search.md +5 -47
- package/rules/memory-stores-durable-facts.md +5 -53
- package/rules/no-contrast-framing.md +5 -64
- package/rules/research-mode.md +5 -25
- package/rules/verify-before-asking.md +5 -48
- package/rules/verify-runtime-state.md +5 -38
|
@@ -0,0 +1,278 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import os
|
|
3
|
+
import subprocess
|
|
4
|
+
import sys
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
import spawn_readiness_hook
|
|
10
|
+
from hooks_constants.spawn_readiness_hook_constants import (
|
|
11
|
+
MISSING_INTERVIEW_REASON,
|
|
12
|
+
MISSING_INVESTIGATION_REASON,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
HOOK_SCRIPT = Path(__file__).resolve().parent / "spawn_readiness_hook.py"
|
|
16
|
+
THREAD_SPAWN_TOOL_NAME = "mcp__hearthbot__start_thread_session"
|
|
17
|
+
SETTLED_LINE = "Scope settled: the request names the file, the fix, and the test."
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _user(text: str) -> dict[str, object]:
|
|
21
|
+
return {"type": "user", "message": {"role": "user", "content": text}}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _tool_use(
|
|
25
|
+
name: str, tool_input: dict[str, object], block_id: str = "toolu_x"
|
|
26
|
+
) -> dict[str, object]:
|
|
27
|
+
return {
|
|
28
|
+
"type": "assistant",
|
|
29
|
+
"message": {
|
|
30
|
+
"role": "assistant",
|
|
31
|
+
"content": [{"type": "tool_use", "id": block_id, "name": name, "input": tool_input}],
|
|
32
|
+
},
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _tool_result(tool_use_id: str, is_error: bool = False) -> dict[str, object]:
|
|
37
|
+
return {
|
|
38
|
+
"type": "user",
|
|
39
|
+
"message": {
|
|
40
|
+
"role": "user",
|
|
41
|
+
"content": [
|
|
42
|
+
{
|
|
43
|
+
"type": "tool_result",
|
|
44
|
+
"tool_use_id": tool_use_id,
|
|
45
|
+
"content": "ok",
|
|
46
|
+
"is_error": is_error,
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
},
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _queued(prompt: str) -> dict[str, object]:
|
|
54
|
+
return {"type": "attachment", "attachment": {"type": "queued_command", "prompt": prompt}}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
READ_STEP = _tool_use("Bash", {"command": "cat README.md"})
|
|
58
|
+
QUESTION_STEP = _tool_use("mcp__hearthbot__ask_decision", {"question": "Which repo should this land in?"})
|
|
59
|
+
REPLY_STEP = _user("The public one.")
|
|
60
|
+
INTERVIEWED_TRANSCRIPT = [_user("Build the hook."), READ_STEP, QUESTION_STEP, REPLY_STEP]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _spawn_input(tool_name: str, brief: str = "Do the task.") -> dict[str, object]:
|
|
64
|
+
if tool_name == THREAD_SPAWN_TOOL_NAME:
|
|
65
|
+
return {"title": "Task", "instructions": brief}
|
|
66
|
+
return {"description": "Task", "prompt": brief}
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _write_transcript(tmp_path: Path, all_entries: list[dict[str, object]]) -> Path:
|
|
70
|
+
transcript_path = tmp_path / "transcript.jsonl"
|
|
71
|
+
transcript_path.write_text(
|
|
72
|
+
"".join(json.dumps(each_entry) + "\n" for each_entry in all_entries), encoding="utf-8"
|
|
73
|
+
)
|
|
74
|
+
return transcript_path
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _run_hook(tmp_path: Path, hook_payload: dict[str, object]) -> str:
|
|
78
|
+
completed = subprocess.run(
|
|
79
|
+
[sys.executable, str(HOOK_SCRIPT)],
|
|
80
|
+
input=json.dumps(
|
|
81
|
+
{"hook_event_name": "PreToolUse", "tool_use_id": "toolu_spawn", **hook_payload}
|
|
82
|
+
),
|
|
83
|
+
capture_output=True,
|
|
84
|
+
text=True,
|
|
85
|
+
env={**os.environ, "HOME": str(tmp_path)},
|
|
86
|
+
check=True,
|
|
87
|
+
)
|
|
88
|
+
return completed.stdout
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _spawn(
|
|
92
|
+
tmp_path: Path,
|
|
93
|
+
all_entries: list[dict[str, object]],
|
|
94
|
+
tool_name: str = "Agent",
|
|
95
|
+
brief: str = "Do the task.",
|
|
96
|
+
) -> str:
|
|
97
|
+
return _run_hook(
|
|
98
|
+
tmp_path,
|
|
99
|
+
{
|
|
100
|
+
"tool_name": tool_name,
|
|
101
|
+
"tool_input": _spawn_input(tool_name, brief),
|
|
102
|
+
"transcript_path": str(_write_transcript(tmp_path, all_entries)),
|
|
103
|
+
},
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _deny_reason(stdout: str) -> str:
|
|
108
|
+
decision = json.loads(stdout)["hookSpecificOutput"]
|
|
109
|
+
assert decision["hookEventName"] == "PreToolUse"
|
|
110
|
+
assert decision["permissionDecision"] == "deny"
|
|
111
|
+
return decision["permissionDecisionReason"]
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _decision_log(tmp_path: Path) -> list[dict[str, object]]:
|
|
115
|
+
log_path = tmp_path / ".claude" / "logs" / "spawn-readiness.jsonl"
|
|
116
|
+
return [
|
|
117
|
+
json.loads(each_line) for each_line in log_path.read_text(encoding="utf-8").splitlines()
|
|
118
|
+
]
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
@pytest.mark.parametrize("tool_name", ["Agent", "Task", THREAD_SPAWN_TOOL_NAME])
|
|
122
|
+
def test_should_let_the_spawn_run_after_a_read_a_question_and_a_reply(
|
|
123
|
+
tmp_path: Path, tool_name: str
|
|
124
|
+
) -> None:
|
|
125
|
+
assert _spawn(tmp_path, INTERVIEWED_TRANSCRIPT, tool_name) == ""
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def test_should_deny_a_spawn_with_no_read_since_the_request(tmp_path: Path) -> None:
|
|
129
|
+
transcript = [_user("Build the hook."), QUESTION_STEP, REPLY_STEP]
|
|
130
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INVESTIGATION_REASON
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def test_should_deny_a_spawn_with_no_question_to_the_user(tmp_path: Path) -> None:
|
|
134
|
+
transcript = [_user("Build the hook."), READ_STEP]
|
|
135
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def test_should_deny_a_spawn_whose_question_has_no_reply_yet(tmp_path: Path) -> None:
|
|
139
|
+
transcript = [_user("Build the hook."), READ_STEP, QUESTION_STEP]
|
|
140
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def test_should_name_both_gaps_when_the_session_did_neither(tmp_path: Path) -> None:
|
|
144
|
+
reason = _deny_reason(_spawn(tmp_path, [_user("Build the hook.")]))
|
|
145
|
+
assert reason == f"{MISSING_INVESTIGATION_REASON} {MISSING_INTERVIEW_REASON}"
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def test_should_count_a_plain_chat_question_as_no_interview(
|
|
149
|
+
tmp_path: Path,
|
|
150
|
+
) -> None:
|
|
151
|
+
statement = _tool_use("mcp__hearthbot__post_message", {"text": "Which repo should this land in?"})
|
|
152
|
+
transcript = [_user("Build the hook."), READ_STEP, statement, REPLY_STEP, READ_STEP]
|
|
153
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def test_should_count_an_answered_ask_user_question_as_the_interview(tmp_path: Path) -> None:
|
|
157
|
+
transcript = [
|
|
158
|
+
_user("Build the hook."),
|
|
159
|
+
_tool_use("Read", {"file_path": "README.md"}),
|
|
160
|
+
_tool_use("AskUserQuestion", {"questions": []}, block_id="toolu_ask"),
|
|
161
|
+
_tool_result("toolu_ask"),
|
|
162
|
+
]
|
|
163
|
+
assert _spawn(tmp_path, transcript) == ""
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def test_should_count_an_answered_widget_as_the_interview(tmp_path: Path) -> None:
|
|
167
|
+
widget = _tool_use("mcp__hearthbot__post_widget", {"family": "visualize", "input": {}})
|
|
168
|
+
transcript = [_user("Build the hook."), READ_STEP, widget, _user("The left layout.")]
|
|
169
|
+
assert _spawn(tmp_path, transcript) == ""
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def test_should_count_an_answered_decision_card_as_the_interview(tmp_path: Path) -> None:
|
|
173
|
+
transcript = [
|
|
174
|
+
_user("Build the hook."),
|
|
175
|
+
_tool_use("mcp__hearthbot__fetch_thread", {"thread_id": "t"}),
|
|
176
|
+
_tool_use("mcp__hearthbot__ask_decision", {"question": "Pick one"}),
|
|
177
|
+
_queued('<wake reason="mention"><message from="human">Option one.</message></wake>'),
|
|
178
|
+
]
|
|
179
|
+
assert _spawn(tmp_path, transcript, THREAD_SPAWN_TOOL_NAME) == ""
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def test_should_skip_an_agent_relay_as_the_reply(tmp_path: Path) -> None:
|
|
183
|
+
relay = _queued('<relay from="coordinator"><note>Keep going.</note></relay>')
|
|
184
|
+
transcript = [_user("Build the hook."), READ_STEP, QUESTION_STEP, relay]
|
|
185
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def test_should_require_a_new_interview_for_a_follow_up_request(tmp_path: Path) -> None:
|
|
189
|
+
earlier_spawn = _tool_use("Agent", {"prompt": "Build the hook."})
|
|
190
|
+
transcript = [*INTERVIEWED_TRANSCRIPT, earlier_spawn, _user("Now add a second hook."), READ_STEP]
|
|
191
|
+
assert _deny_reason(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def test_should_treat_two_messages_in_a_row_after_a_question_as_one_reply(
|
|
195
|
+
tmp_path: Path,
|
|
196
|
+
) -> None:
|
|
197
|
+
transcript = [*INTERVIEWED_TRANSCRIPT, _user("Also keep it short.")]
|
|
198
|
+
assert _spawn(tmp_path, transcript) == ""
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def test_should_pass_the_interview_on_a_scope_settled_line_and_log_it(tmp_path: Path) -> None:
|
|
202
|
+
transcript = [_user("Fix the typo in README.md line 3."), READ_STEP]
|
|
203
|
+
brief = f"Fix the typo.\n{SETTLED_LINE}"
|
|
204
|
+
assert _spawn(tmp_path, transcript, THREAD_SPAWN_TOOL_NAME, brief) == ""
|
|
205
|
+
[log_record] = _decision_log(tmp_path)
|
|
206
|
+
assert log_record["outcome"] == "scope_settled"
|
|
207
|
+
assert log_record["tool_name"] == THREAD_SPAWN_TOOL_NAME
|
|
208
|
+
assert log_record["scope_settled_line"] == SETTLED_LINE
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def test_should_still_require_the_read_with_a_scope_settled_line(tmp_path: Path) -> None:
|
|
212
|
+
brief = f"Fix the typo.\n{SETTLED_LINE}"
|
|
213
|
+
reason = _deny_reason(_spawn(tmp_path, [_user("Fix the typo.")], brief=brief))
|
|
214
|
+
assert reason == MISSING_INVESTIGATION_REASON
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def test_should_deny_a_scope_settled_line_with_no_reason(tmp_path: Path) -> None:
|
|
218
|
+
transcript = [_user("Fix the typo."), READ_STEP]
|
|
219
|
+
reason = _deny_reason(_spawn(tmp_path, transcript, brief="Fix it.\nScope settled:"))
|
|
220
|
+
assert reason == MISSING_INTERVIEW_REASON
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def test_should_let_a_read_only_subagent_run_without_checks(tmp_path: Path) -> None:
|
|
224
|
+
stdout = _run_hook(
|
|
225
|
+
tmp_path,
|
|
226
|
+
{
|
|
227
|
+
"tool_name": "Agent",
|
|
228
|
+
"tool_input": {"subagent_type": "Explore", "prompt": "Find the hooks."},
|
|
229
|
+
"transcript_path": str(_write_transcript(tmp_path, [_user("Build the hook.")])),
|
|
230
|
+
},
|
|
231
|
+
)
|
|
232
|
+
assert stdout == ""
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def test_should_count_a_read_only_subagent_as_the_investigation(tmp_path: Path) -> None:
|
|
236
|
+
explore = _tool_use("Agent", {"subagent_type": "Explore", "prompt": "Find it."})
|
|
237
|
+
transcript = [_user("Build the hook."), explore, QUESTION_STEP, REPLY_STEP]
|
|
238
|
+
assert _spawn(tmp_path, transcript) == ""
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def test_should_let_a_spawn_from_inside_a_subagent_run(tmp_path: Path) -> None:
|
|
242
|
+
stdout = _run_hook(
|
|
243
|
+
tmp_path,
|
|
244
|
+
{
|
|
245
|
+
"tool_name": "Agent",
|
|
246
|
+
"agent_id": "agent_1",
|
|
247
|
+
"tool_input": _spawn_input("Agent"),
|
|
248
|
+
"transcript_path": str(_write_transcript(tmp_path, [_user("Build the hook.")])),
|
|
249
|
+
},
|
|
250
|
+
)
|
|
251
|
+
assert stdout == ""
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def test_should_let_the_spawn_run_and_log_when_the_transcript_is_unreadable(
|
|
255
|
+
tmp_path: Path,
|
|
256
|
+
) -> None:
|
|
257
|
+
stdout = _run_hook(
|
|
258
|
+
tmp_path,
|
|
259
|
+
{
|
|
260
|
+
"tool_name": "Agent",
|
|
261
|
+
"tool_input": _spawn_input("Agent"),
|
|
262
|
+
"transcript_path": str(tmp_path / "absent.jsonl"),
|
|
263
|
+
},
|
|
264
|
+
)
|
|
265
|
+
assert stdout == ""
|
|
266
|
+
[log_record] = _decision_log(tmp_path)
|
|
267
|
+
assert log_record["outcome"] == "transcript_unreadable"
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def test_should_return_the_scope_settled_line_from_the_brief() -> None:
|
|
271
|
+
tool_input = {"instructions": f"Do X.\n {SETTLED_LINE}"}
|
|
272
|
+
assert (
|
|
273
|
+
spawn_readiness_hook.scope_settled_line(THREAD_SPAWN_TOOL_NAME, tool_input) == SETTLED_LINE
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def test_should_ignore_a_tool_it_does_not_check() -> None:
|
|
278
|
+
assert spawn_readiness_hook.decide_hook_output({"tool_name": "Read", "tool_input": {}}) is None
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
from spawn_readiness_steps import SessionStep, readiness_gaps, session_steps
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def _user(text: str) -> dict[str, object]:
|
|
7
|
+
return {"type": "user", "message": {"role": "user", "content": text}}
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _tool_use(name: str, tool_input: dict[str, object], block_id: str = "toolu_x") -> dict[str, object]:
|
|
11
|
+
return {
|
|
12
|
+
"type": "assistant",
|
|
13
|
+
"message": {
|
|
14
|
+
"role": "assistant",
|
|
15
|
+
"content": [{"type": "tool_use", "id": block_id, "name": name, "input": tool_input}],
|
|
16
|
+
},
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _tool_result(tool_use_id: str) -> dict[str, object]:
|
|
21
|
+
return {
|
|
22
|
+
"type": "user",
|
|
23
|
+
"message": {
|
|
24
|
+
"role": "user",
|
|
25
|
+
"content": [{"type": "tool_result", "tool_use_id": tool_use_id, "content": "ok"}],
|
|
26
|
+
},
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
READ_STEP = _tool_use("Bash", {"command": "cat README.md"})
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_should_skip_meta_entries_and_tool_results_as_user_messages() -> None:
|
|
34
|
+
all_lines = [
|
|
35
|
+
json.dumps(_user("Build the hook.")),
|
|
36
|
+
json.dumps({**_user("Base directory for this skill"), "isMeta": True}),
|
|
37
|
+
json.dumps(READ_STEP),
|
|
38
|
+
json.dumps(_tool_result("toolu_x")),
|
|
39
|
+
]
|
|
40
|
+
assert session_steps(all_lines) == [
|
|
41
|
+
SessionStep.USER_MESSAGE,
|
|
42
|
+
SessionStep.READ,
|
|
43
|
+
]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def test_should_leave_the_spawn_under_check_out_of_the_steps() -> None:
|
|
47
|
+
explore = _tool_use(
|
|
48
|
+
"Agent", {"subagent_type": "Explore", "prompt": "Find it."}, block_id="toolu_spawn"
|
|
49
|
+
)
|
|
50
|
+
all_lines = [json.dumps(_user("Build the hook.")), json.dumps(explore)]
|
|
51
|
+
assert session_steps(all_lines, "toolu_spawn") == [
|
|
52
|
+
SessionStep.USER_MESSAGE
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def test_should_report_no_gaps_for_a_reply_after_a_question() -> None:
|
|
57
|
+
all_steps = [
|
|
58
|
+
SessionStep.USER_MESSAGE,
|
|
59
|
+
SessionStep.READ,
|
|
60
|
+
SessionStep.QUESTION,
|
|
61
|
+
SessionStep.USER_MESSAGE,
|
|
62
|
+
]
|
|
63
|
+
assert readiness_gaps(all_steps) == []
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def test_should_read_a_github_get_call_as_a_read_and_a_chat_question_as_another_call() -> None:
|
|
67
|
+
all_lines = [
|
|
68
|
+
json.dumps(_tool_use("mcp__github__get_file_contents", {"path": "a"})),
|
|
69
|
+
json.dumps(_tool_use("mcp__hearthbot__reply", {"text": "Which one?"})),
|
|
70
|
+
json.dumps(_tool_use("mcp__hearthbot__ask_decision", {"question": "Which one?"})),
|
|
71
|
+
]
|
|
72
|
+
assert session_steps(all_lines) == [SessionStep.READ, SessionStep.OTHER_TOOL_CALL, SessionStep.QUESTION]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def test_should_start_the_span_at_the_request_before_an_answered_question() -> None:
|
|
76
|
+
all_steps = [
|
|
77
|
+
SessionStep.USER_MESSAGE,
|
|
78
|
+
SessionStep.READ,
|
|
79
|
+
SessionStep.OTHER_TOOL_CALL,
|
|
80
|
+
SessionStep.USER_MESSAGE,
|
|
81
|
+
SessionStep.QUESTION,
|
|
82
|
+
SessionStep.USER_MESSAGE,
|
|
83
|
+
]
|
|
84
|
+
assert readiness_gaps(all_steps) == [
|
|
85
|
+
"Investigate the request before this spawn. Read the files, threads, or sources it names, "
|
|
86
|
+
"so the brief and the agent count fit the task. Run the reads in a message before the spawn."
|
|
87
|
+
]
|
package/package.json
CHANGED
|
@@ -1,40 +1,9 @@
|
|
|
1
|
-
# ASD-STE100
|
|
1
|
+
# ASD-STE100 language policy
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
It defines ordinary word choice, sentence style, tone, punctuation, and prose form.
|
|
3
|
+
**When:** Writing chat, tool narration, or repository prose.
|
|
5
4
|
|
|
6
|
-
|
|
7
|
-
clear-writing principles to assistant messages. The official standard remains the authority for
|
|
8
|
-
definitions and dictionary decisions.
|
|
5
|
+
Use the sole general language rule. Write short, complete sentences on one topic. Use active voice; lead with conditions; give one action per step. Use plain, precise words and stable terms. Expand abbreviations and contractions first; name unclear pronouns. Be inclusive; punctuate clearly. Preserve exact labels, identifiers, formulas, titles, and interface text. Send a result, blocker, or question. Mark injury or death `WARNING`, equipment damage `CAUTION`; state condition then result. Aim for 20 words per step, 25 per description. Treat checks as aids; have a human verify accuracy, terms, safety, confidentiality, and meaning.
|
|
9
6
|
|
|
10
|
-
|
|
7
|
+
**Enforcement:** none, the agent applies it.
|
|
11
8
|
|
|
12
|
-
|
|
13
|
-
- [ASD-STE100 FAQ](https://www.asd-ste100.org/STE_faq.html)
|
|
14
|
-
- [ASD-STE100 About page](https://www.asd-ste100.org/about_STE.html)
|
|
15
|
-
- [ASD-STE100 tools guidance](https://www.asd-ste100.org/STEsoftware.html)
|
|
16
|
-
- [STEMG white paper on ASD-STE100 and artificial intelligence](https://www.asd-ste100.org/assets/files/WhitePaper-ASD-STE100_and_AI.pdf)
|
|
17
|
-
|
|
18
|
-
## Writing policy
|
|
19
|
-
|
|
20
|
-
- Write short, complete sentences. Keep one topic in each explanatory sentence.
|
|
21
|
-
- Use active voice. Put the condition first when an instruction has a prerequisite.
|
|
22
|
-
- Write imperative procedure steps. Give one action in each sentence.
|
|
23
|
-
- Prefer familiar, precise words. Use one stable term for one item or action.
|
|
24
|
-
- Write full words and explicit references. Expand abbreviations and contractions when they first appear.
|
|
25
|
-
- Replace ambiguous pronouns with the noun they name.
|
|
26
|
-
- Use inclusive, neutral language.
|
|
27
|
-
- Use periods, commas, colons, and bullets to show structure.
|
|
28
|
-
- Preserve exact quoted labels, identifiers, formulas, titles, and interface text.
|
|
29
|
-
- Send the reader only a result, a blocker, or a question. Leave out a line that says nothing is needed from them, and leave out which agent, session, or coordinator did the work.
|
|
30
|
-
- Use `WARNING` for a risk of injury or death. Use `CAUTION` for a risk of equipment, tool, or machine damage. State the command or condition first, then state the result.
|
|
31
|
-
- Aim for 20 words or fewer in a procedure sentence when the technical content allows.
|
|
32
|
-
- Aim for 25 words or fewer in a descriptive sentence when the technical content allows.
|
|
33
|
-
|
|
34
|
-
Use this policy for chat, tool narration, questions, plans, documentation, code-adjacent prose,
|
|
35
|
-
and durable repository text. Named contracts can add behavior-specific structure, evidence,
|
|
36
|
-
question routing, current-state documentation, completion, docstring, or publication rules.
|
|
37
|
-
Those contracts use this policy for their language.
|
|
38
|
-
|
|
39
|
-
Treat automated output and language checks as drafting aids. A responsible human verifies
|
|
40
|
-
technical accuracy, terminology, safety, confidentiality, and intended meaning.
|
|
9
|
+
**Full text:** [`docs/rule-guides/asd-ste100-language.md`](../docs/rule-guides/asd-ste100-language.md). Read it for sources.
|
|
@@ -1,33 +1,9 @@
|
|
|
1
|
-
# Clean
|
|
1
|
+
# Clean up temporary files
|
|
2
2
|
|
|
3
|
-
**When
|
|
3
|
+
**When:** Creating scratch files, debug dumps, or one-off helpers during a task.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
Prefer memory to scratch files and track any temporary files you create. At completion, remove those files and leave user-requested files in place. Files under the OS temporary root or `$CLAUDE_JOB_DIR` need no explicit removal; a parent handles child-agent scratch. Use the permitted removal form.
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
**Enforcement:** none, the agent applies it.
|
|
8
8
|
|
|
9
|
-
|
|
10
|
-
- When a temporary file is needed (e.g., a helper script, a test fixture, a debug output), track it mentally for cleanup.
|
|
11
|
-
|
|
12
|
-
## When a task is complete
|
|
13
|
-
|
|
14
|
-
- Remove every temporary file, script, or helper file you created during the task.
|
|
15
|
-
- Leave the working directory cleaner than you found it.
|
|
16
|
-
- If a file was created at the user's explicit request (not as a byproduct of your process), leave it in place.
|
|
17
|
-
|
|
18
|
-
## Exceptions to the removal duty
|
|
19
|
-
|
|
20
|
-
Three kinds of file are already ephemeral and need no explicit removal:
|
|
21
|
-
|
|
22
|
-
- A file under the OS temporary root.
|
|
23
|
-
- A file under `$CLAUDE_JOB_DIR`, which the harness clears with the job.
|
|
24
|
-
- A child agent's scratch file, which the parent removes at teardown.
|
|
25
|
-
|
|
26
|
-
Use an allowed removal form for everything else: [`destructive-commands.md`](destructive-commands.md) names them.
|
|
27
|
-
|
|
28
|
-
## What counts as temporary
|
|
29
|
-
|
|
30
|
-
- Scripts written to test a hypothesis or run a one-off check
|
|
31
|
-
- Debug output files, log dumps, or intermediate data exports
|
|
32
|
-
- Helper files created to work around tool limitations
|
|
33
|
-
- Any file the user did not ask for and would not expect to find after the task
|
|
9
|
+
**Full text:** [`docs/rule-guides/cleanup-temp-files.md`](../docs/rule-guides/cleanup-temp-files.md). Read it to classify a file or choose cleanup.
|
package/rules/correction-lens.md
CHANGED
|
@@ -1,110 +1,21 @@
|
|
|
1
|
-
# Correction
|
|
1
|
+
# Correction lens
|
|
2
2
|
|
|
3
|
-
**
|
|
4
|
-
clause at the end of this document states what that binds.
|
|
3
|
+
**When:** A user corrects you.
|
|
5
4
|
|
|
6
|
-
|
|
7
|
-
"stop writing that word", a repeated review comment, a fix the user makes by hand
|
|
8
|
-
after an agent hands work back. Each one is evidence that a control is missing,
|
|
9
|
-
and the correction is handled by building that control.
|
|
5
|
+
Open a control in the guarded repository, in the same run, at the highest capable layer. State the layer, why higher layers fail, and the change opened. Pair layers when both help. Move repeated corrections up a layer. Memory records the decision.
|
|
10
6
|
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
|
18
|
-
|---|---|---|---|
|
|
19
|
-
| 1 | Codebase | The mistake is impossible by how the code is written | A type, a signature, a data structure, an API shape, a deleted branch |
|
|
20
|
-
| 2 | Static analysis | A program reads the tree and decides | A lint rule, an enforcer check, a paired test, a CI gate |
|
|
21
|
-
| 3 | Review tooling | A reviewer or a review bot reads the criterion | A `CODE_RULES.md` row, a `.cursor/BUGBOT.md` pointer, a Graphite or BugBot rule |
|
|
22
|
-
| 4 | Skill | An agent follows a procedure | A skill under the agents home |
|
|
23
|
-
| 5 | Style guide | Word choice and prose shape | A row in a `rules/*.md` file or a style document |
|
|
24
|
-
|
|
25
|
-
Layer 1 is the goal every time. A lesson encoded there needs no reader, no run,
|
|
26
|
-
and no agent to remember it. The wrong call has nowhere to live.
|
|
27
|
-
|
|
28
|
-
## Choosing the layer
|
|
29
|
-
|
|
30
|
-
Three sentences close every correction:
|
|
31
|
-
|
|
32
|
-
1. The layer chosen.
|
|
33
|
-
2. Why each higher layer cannot hold this lesson.
|
|
34
|
-
3. The change opened at the chosen layer, by pull request.
|
|
35
|
-
|
|
36
|
-
A layer holds a lesson when its own test passes:
|
|
37
|
-
|
|
38
|
-
- **Layer 1 holds it** when a type, a signature, or a data structure can make the
|
|
39
|
-
wrong call fail to compile, fail to construct, or fail to exist.
|
|
40
|
-
- **Layer 2 holds it** when a program reading the tree can separate right from
|
|
41
|
-
wrong with no judgment call.
|
|
42
|
-
- **Layer 3 holds it** when a reader needs the criterion in front of them and a
|
|
43
|
-
program cannot decide it.
|
|
44
|
-
- **Layer 4 holds it** when the lesson is a procedure with steps an agent runs.
|
|
45
|
-
- **Layer 5 holds it** when the lesson is word choice or the shape of prose.
|
|
46
|
-
|
|
47
|
-
Effort rules out no layer. A layer is ruled out when it lacks the capability to
|
|
48
|
-
hold the lesson, and the sentence that rules it out says which capability is
|
|
49
|
-
missing. "A lint for this would take a day" leaves layer 2 in play.
|
|
50
|
-
|
|
51
|
-
Two layers often hold one correction. Take both: the shape at layer 1 and the
|
|
52
|
-
check at layer 2 cost one run together and each covers what the other misses.
|
|
53
|
-
|
|
54
|
-
## A repeated correction moves up a layer
|
|
55
|
-
|
|
56
|
-
The same correction arriving a second time is the measurement. The layer chosen
|
|
57
|
-
the first time was too low. Move the lesson one layer up and say so in the same
|
|
58
|
-
run.
|
|
59
|
-
|
|
60
|
-
A review comment that repeats across pull requests reads the same way. Each
|
|
61
|
-
repetition names the layer below the reviewer as the one that needs the control.
|
|
62
|
-
Count the repetitions.
|
|
63
|
-
|
|
64
|
-
## Where the control lands
|
|
65
|
-
|
|
66
|
-
A control lands in the repository whose code, CI, or pipeline it guards. A
|
|
67
|
-
control for one repository's workflows, review bots, pipelines, or agents lands
|
|
68
|
-
in that repository. This package takes only repo-agnostic controls, the
|
|
69
|
-
environment and development pieces any repository could use, so every session
|
|
70
|
-
that loads this environment carries them. A control written into one project's
|
|
71
|
-
instructions reaches that project alone, and a control written into a chat reply
|
|
72
|
-
reaches that conversation alone.
|
|
73
|
-
|
|
74
|
-
Memory holds the decision. The control holds the behavior. A correction that
|
|
75
|
-
produced a memory file and nothing else has been recorded and never encoded, so
|
|
76
|
-
the rule still binds. Open the control.
|
|
77
|
-
|
|
78
|
-
## Trust grows as the controls catch the common mistakes
|
|
79
|
-
|
|
80
|
-
Add agents after the controls catch what agents get wrong, and start from a
|
|
81
|
-
workflow small enough to watch. Each repeated correction becomes a stronger
|
|
82
|
-
control, and the number of agents rises behind it.
|
|
83
|
-
[`docs/high-trust-agent-delivery.md`](../docs/high-trust-agent-delivery.md)
|
|
84
|
-
carries the model this comes from, including the layered-controls picture and
|
|
85
|
-
the inner and outer delivery loops.
|
|
7
|
+
| Priority | Layer | Control |
|
|
8
|
+
|---|---|---|
|
|
9
|
+
| 1 | Codebase | Prevent the mistake. |
|
|
10
|
+
| 2 | Static analysis | Programmatic check. |
|
|
11
|
+
| 3 | Review tooling | Reviewer criterion. |
|
|
12
|
+
| 4 | Skill | Agent procedure. |
|
|
13
|
+
| 5 | Style guide | Wording rule. |
|
|
86
14
|
|
|
87
15
|
## Permanence
|
|
88
16
|
|
|
89
|
-
|
|
90
|
-
a rewrite that would move it into `rules-archived/` stops at this line. The
|
|
91
|
-
"Never archived" section of
|
|
92
|
-
[`packages/claude-dev-env/rules-archived/ARCHIVE-MANIFEST.md`](../rules-archived/ARCHIVE-MANIFEST.md)
|
|
93
|
-
names it, and
|
|
94
|
-
[`archiving-agent-config.md`](archiving-agent-config.md) names the same
|
|
95
|
-
exemption from the archiving procedure's side.
|
|
96
|
-
|
|
97
|
-
An edit that sharpens this rule is welcome. The file stays.
|
|
98
|
-
|
|
99
|
-
## Codex copy
|
|
100
|
-
|
|
101
|
-
Codex reads its repository `AGENTS.md`. The standalone excerpt it receives lives in [`docs/rule-guides/correction-lens-excerpt.md`](../docs/rule-guides/correction-lens-excerpt.md).
|
|
17
|
+
Keep this file in `rules/`. A prune, archive, consolidation, or rewrite that would move it stops here. Edits that sharpen it are welcome.
|
|
102
18
|
|
|
103
|
-
|
|
19
|
+
**Enforcement:** none, the agent applies it.
|
|
104
20
|
|
|
105
|
-
|
|
106
|
-
|---|---|
|
|
107
|
-
| [`flag-non-breaking-findings.md`](flag-non-breaking-findings.md) | A gate blocks on a breaking finding and records a smell |
|
|
108
|
-
| [`code-standards.md`](code-standards.md) | The layer map of contract, pointer, enforcer, lint, session rules |
|
|
109
|
-
| [`archiving-agent-config.md`](archiving-agent-config.md) | How a rule leaves service, and which files are exempt |
|
|
110
|
-
| [`falsify-before-green.md`](falsify-before-green.md) | A new layer-2 check counts once it has run red on a named break |
|
|
21
|
+
**Full text:** [`docs/rule-guides/correction-lens.md`](../docs/rule-guides/correction-lens.md). Read it when choosing a layer or locating a control.
|
|
@@ -1,47 +1,11 @@
|
|
|
1
|
-
# Destructive
|
|
1
|
+
# Destructive commands in Bash
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
**When:** Removing files or writing a destructive command string.
|
|
4
4
|
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
## Removal forms to prefer
|
|
8
|
-
|
|
9
|
-
- **Scratch and probe files.** Use the PowerShell tool with `Remove-Item -Recurse -Force -Confirm:$false <absolute path>`.
|
|
10
|
-
- **Worktrees.** Use `git worktree remove --force <path>`.
|
|
11
|
-
- **Tracked files.** Use `git rm <path>`, which records the deletion in the index.
|
|
12
|
-
- **Bash `rm` when unavoidable.** Write one standalone `rm` with absolute literal paths, no chaining, and no globs. Keep every target inside the ephemeral namespace below.
|
|
13
|
-
|
|
14
|
-
## The ephemeral namespace
|
|
15
|
-
|
|
16
|
-
Keep a Bash `rm` to targets that resolve inside one of these:
|
|
17
|
-
|
|
18
|
-
- The OS temporary root.
|
|
19
|
-
- A path rooted at `/tmp` or `/temp`, drive-letter tolerant.
|
|
20
|
-
- A path holding a `/worktrees/` or `/worktree/` segment, or a directory git reports inside a worktree admin directory.
|
|
21
|
-
- `~/.claude`.
|
|
22
|
-
|
|
23
|
-
Never pass a bare ephemeral root, such as `/tmp`, the OS temp root itself, or a bare directory named `worktrees` or `worktree`. A single stray argument then wipes the whole namespace.
|
|
24
|
-
|
|
25
|
-
Write each target as a literal path. A variable, a `$(...)` or backtick expansion, or a brace glob hides what the command will delete from the reader and from the permission matcher.
|
|
26
|
-
|
|
27
|
-
A file left in the OS temp directory or under `$CLAUDE_JOB_DIR` is cleaned by the harness and needs no explicit removal. See the exception clause in [`cleanup-temp-files.md`](cleanup-temp-files.md).
|
|
28
|
-
|
|
29
|
-
## Keep destructive literals out of the command string
|
|
30
|
-
|
|
31
|
-
The permission matcher reads the raw command string. A destructive literal carried only as data still sits in that string, so it can push the command out of an allowed shape and into a prompt even though the shell never executes it. This covers a commit message, a PR or issue body, an echoed string, a `python -c` or `node -e` or `awk` argument, and a heredoc.
|
|
32
|
-
|
|
33
|
-
- Bodies that describe destructive-command behavior go in a file passed by path, such as `git commit -F <file>` or `gh … --body-file <file>`. See [`gh-cli-conventions.md`](gh-cli-conventions.md). Never `git commit -m` or `gh … -b`.
|
|
34
|
-
- To exercise or verify a hook, run the committed test suite with `python -m pytest <test_file>`, which passes the command strings as in-language data. Never an inline `python -c` harness.
|
|
35
|
-
|
|
36
|
-
## Every subagent prompt carries the rule
|
|
37
|
-
|
|
38
|
-
A prompt-delivered directive reaches only the agent that gets it. An agent that spawns its own workers, such as review lenses, fix agents, or verifiers, copies this line into every subagent prompt it issues. A grandchild cleaning up its own probe file then uses an allowed form:
|
|
5
|
+
No hook watches Bash commands; a permission prompt can stall unattended work. Keep destructive literals out of command strings, including data. Use literal absolute targets; never target a bare temporary root. Use `git rm` for tracked files. Pass bodies by file and test hooks through the test suite. Copy this line into every subagent prompt:
|
|
39
6
|
|
|
40
7
|
> Never use bash rm in any form. Delete scratch/probe files with the PowerShell tool (Remove-Item -Recurse -Force -Confirm:$false <absolute path>), or leave them in the OS temp dir; remove worktrees only via git worktree remove --force.
|
|
41
8
|
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
## Sibling rules
|
|
9
|
+
**Enforcement:** none, the agent applies it; the harness may prompt.
|
|
45
10
|
|
|
46
|
-
|
|
47
|
-
- [`windows-filesystem-safe.md`](windows-filesystem-safe.md) holds the safe `rmtree` and `force_rmtree` patterns for read-only Windows files.
|
|
11
|
+
**Full text:** [`docs/rule-guides/destructive-commands.md`](../docs/rule-guides/destructive-commands.md). Read it before any removal.
|