claude-dev-env 8.40.1 → 8.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/agents/test_agent_frontmatter.py +17 -14
- package/.agents/agents-archived/clean-coder.md +4 -4
- package/.agents/agents-archived/pr-description-writer.md +1 -1
- package/.agents/skills/pr-lifecycle/SKILL.md +470 -0
- package/.agents/skills/privacy-hygiene/reference/sweep-procedure.md +1 -1
- package/bin/ever-shipped-skills.mjs +1 -0
- package/bin/ever-shipped-skills.test.mjs +4 -0
- package/bin/install.cursor-rules.test.mjs +2 -1
- package/docs/rule-guides/code-standards.md +1 -1
- package/docs/rule-guides/destructive-commands.md +1 -1
- package/hooks/blocking/pr_lifecycle_skill_gate.py +190 -0
- package/hooks/blocking/test_pr_lifecycle_skill_gate.py +163 -0
- package/hooks/hooks.json +25 -0
- package/hooks/hooks_constants/pr_lifecycle_skill_gate_constants.py +33 -0
- package/hooks/hooks_constants/skill_loaded_reminder_constants.py +0 -10
- package/hooks/hooks_constants/spawn_readiness_hook_constants.py +30 -12
- package/hooks/hooks_constants/test_pr_lifecycle_skill_gate_constants.py +12 -0
- package/hooks/hooks_constants/test_transcript_skill_scan_constants.py +7 -0
- package/hooks/hooks_constants/transcript_skill_scan_constants.py +7 -0
- package/hooks/routing/spawn_readiness_hook.py +118 -46
- package/hooks/routing/test_spawn_readiness_hook.py +113 -21
- package/hooks/routing/test_spawn_readiness_steps.py +2 -4
- package/hooks/session/skill_loaded_reminder.py +4 -51
- package/hooks/session/test_skill_loaded_reminder.py +4 -0
- package/hooks/test_transcript_skill_scan.py +58 -0
- package/hooks/transcript_skill_scan.py +93 -0
- package/package.json +1 -1
- package/rules/flag-non-breaking-findings.md +2 -2
- package/rules/shell-invocation.md +2 -14
- package/rules/skill-pointers.md +3 -0
- package/scripts/policy_lint/adapter_configuration.py +5 -0
- package/scripts/policy_lint/config/constants.py +1 -0
- package/scripts/tests/test_adapter_configuration.py +8 -0
- package/scripts/tests/test_banned_prose_words.py +1 -1
- package/scripts/tests/test_rule_load_scopes.py +1 -7
- package/rules/agent-merges-its-own-green-pull-request.md +0 -55
- package/rules/ci-owns-the-gate.md +0 -101
- package/rules/durable-post-artifacts.md +0 -76
- package/rules/gh-cli-conventions.md +0 -36
- package/rules/git-workflow.md +0 -121
- package/rules/re-stage-before-commit.md +0 -14
- package/rules/review-closure-is-a-check.md +0 -54
|
@@ -1,20 +1,24 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
|
-
"""PreToolUse hook:
|
|
2
|
+
"""PreToolUse hook: remind a session to read and ask before it spawns agents.
|
|
3
3
|
|
|
4
|
-
Registered on ``Agent``, ``Task``,
|
|
5
|
-
|
|
4
|
+
Registered on ``Agent``, ``Task``, ``mcp__hearthbot__start_thread_session``,
|
|
5
|
+
``multi_agent_v1__spawn_agent``, ``Workflow``, and a workflow dispatch whose
|
|
6
|
+
inputs carry a ``prompt``. It reads the session transcript and checks two
|
|
7
|
+
things since the user's request:
|
|
6
8
|
|
|
7
9
|
::
|
|
8
10
|
|
|
9
|
-
no read step after the request ->
|
|
10
|
-
no interactive question, then an answer ->
|
|
11
|
+
no read step after the request -> context: investigate first
|
|
12
|
+
no interactive question, then an answer -> context: interview first
|
|
11
13
|
brief line "Scope settled: <reason>" -> interview check passes, logged
|
|
12
|
-
both found -> no output
|
|
14
|
+
both found -> no output
|
|
15
|
+
Codex payload (turn_id), no settled line -> context: the Codex reminder
|
|
13
16
|
|
|
14
|
-
``
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
17
|
+
The output is ``additionalContext`` alone, so the spawn runs and keeps its
|
|
18
|
+
normal permission flow. ``spawn_readiness_steps`` reads the transcript and
|
|
19
|
+
finds the request. Four cases pass without a check: a call from inside a
|
|
20
|
+
subagent, a read-only subagent type, a tool input that is not an object, and a
|
|
21
|
+
transcript the hook cannot read. The last one is logged.
|
|
18
22
|
"""
|
|
19
23
|
|
|
20
24
|
from __future__ import annotations
|
|
@@ -31,13 +35,19 @@ routing_directory = str(Path(__file__).resolve().parent)
|
|
|
31
35
|
if routing_directory not in sys.path:
|
|
32
36
|
sys.path.insert(0, routing_directory)
|
|
33
37
|
|
|
34
|
-
from hooks_constants.hook_block_logger import log_hook_block
|
|
35
38
|
from hooks_constants.pre_tool_use_stdin import read_hook_input_dictionary_from_stdin
|
|
36
39
|
from hooks_constants.spawn_readiness_hook_constants import (
|
|
40
|
+
ADDITIONAL_CONTEXT_KEY,
|
|
37
41
|
AGENT_ID_KEY,
|
|
38
42
|
ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME,
|
|
39
43
|
ALL_READ_ONLY_SUBAGENT_TYPES,
|
|
44
|
+
CODEX_TRANSCRIPT_REMINDER,
|
|
45
|
+
CODEX_TURN_ID_KEY,
|
|
40
46
|
DECISION_LOG_RELATIVE_PATH,
|
|
47
|
+
DISPATCH_INPUTS_KEY,
|
|
48
|
+
DISPATCH_METHOD_INPUT_KEY,
|
|
49
|
+
DISPATCH_PROMPT_INPUT_KEY,
|
|
50
|
+
DISPATCH_RUN_WORKFLOW_METHOD,
|
|
41
51
|
HOOK_EVENT_NAME_KEY,
|
|
42
52
|
HOOK_SPECIFIC_OUTPUT_KEY,
|
|
43
53
|
LOG_APPEND_MODE,
|
|
@@ -48,11 +58,9 @@ from hooks_constants.spawn_readiness_hook_constants import (
|
|
|
48
58
|
LOG_TOOL_NAME_KEY,
|
|
49
59
|
LOG_TOOL_USE_ID_KEY,
|
|
50
60
|
MISSING_INTERVIEW_REASON,
|
|
61
|
+
OUTCOME_REMINDED,
|
|
51
62
|
OUTCOME_SCOPE_SETTLED,
|
|
52
63
|
OUTCOME_TRANSCRIPT_UNREADABLE,
|
|
53
|
-
PERMISSION_DECISION_KEY,
|
|
54
|
-
PERMISSION_DECISION_REASON_KEY,
|
|
55
|
-
PERMISSION_DENY,
|
|
56
64
|
PRE_TOOL_USE_EVENT_NAME,
|
|
57
65
|
REASON_SEPARATOR,
|
|
58
66
|
SCOPE_SETTLED_PREFIX,
|
|
@@ -63,11 +71,67 @@ from hooks_constants.spawn_readiness_hook_constants import (
|
|
|
63
71
|
TRANSCRIPT_DECODE_ERRORS,
|
|
64
72
|
TRANSCRIPT_ENCODING,
|
|
65
73
|
TRANSCRIPT_PATH_KEY,
|
|
74
|
+
WORKFLOW_DISPATCH_TOOL_NAME,
|
|
75
|
+
WORKFLOW_SCRIPT_PATH_INPUT_KEY,
|
|
76
|
+
WORKFLOW_TOOL_NAME,
|
|
66
77
|
)
|
|
67
78
|
from spawn_readiness_steps import readiness_gaps, session_steps
|
|
68
79
|
|
|
69
80
|
|
|
70
|
-
def
|
|
81
|
+
def _workflow_script(all_tool_input_fields: dict[str, object]) -> str:
|
|
82
|
+
script = all_tool_input_fields.get(ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME[WORKFLOW_TOOL_NAME])
|
|
83
|
+
if isinstance(script, str):
|
|
84
|
+
return script
|
|
85
|
+
script_path = all_tool_input_fields.get(WORKFLOW_SCRIPT_PATH_INPUT_KEY)
|
|
86
|
+
if not isinstance(script_path, str) or not script_path:
|
|
87
|
+
return ""
|
|
88
|
+
try:
|
|
89
|
+
return Path(script_path).read_text(
|
|
90
|
+
encoding=TRANSCRIPT_ENCODING, errors=TRANSCRIPT_DECODE_ERRORS
|
|
91
|
+
)
|
|
92
|
+
except OSError:
|
|
93
|
+
return ""
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _dispatch_prompt(all_tool_input_fields: dict[str, object]) -> str | None:
|
|
97
|
+
if all_tool_input_fields.get(DISPATCH_METHOD_INPUT_KEY) != DISPATCH_RUN_WORKFLOW_METHOD:
|
|
98
|
+
return None
|
|
99
|
+
workflow_inputs = all_tool_input_fields.get(DISPATCH_INPUTS_KEY)
|
|
100
|
+
if not isinstance(workflow_inputs, dict):
|
|
101
|
+
return None
|
|
102
|
+
prompt = workflow_inputs.get(DISPATCH_PROMPT_INPUT_KEY)
|
|
103
|
+
return prompt if isinstance(prompt, str) else None
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def spawn_brief(tool_name: str, all_tool_input_fields: dict[str, object]) -> str | None:
|
|
107
|
+
"""Return the brief a spawn call hands its agents, or None when the call spawns none.
|
|
108
|
+
|
|
109
|
+
::
|
|
110
|
+
|
|
111
|
+
Agent {"prompt": "Do X."} -> "Do X."
|
|
112
|
+
Workflow {"scriptPath": "w.js"} -> the text of w.js
|
|
113
|
+
actions trigger {"method": "run_workflow",
|
|
114
|
+
"inputs": {"prompt": "Do X."}} -> "Do X."
|
|
115
|
+
actions trigger {"method": "cancel_workflow_run"} -> None
|
|
116
|
+
Agent {"subagent_type": "Explore", "prompt": ...} -> None
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
tool_name: The tool the session called.
|
|
120
|
+
all_tool_input_fields: The tool input.
|
|
121
|
+
"""
|
|
122
|
+
if tool_name == WORKFLOW_DISPATCH_TOOL_NAME:
|
|
123
|
+
return _dispatch_prompt(all_tool_input_fields)
|
|
124
|
+
if tool_name not in ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME:
|
|
125
|
+
return None
|
|
126
|
+
if all_tool_input_fields.get(SUBAGENT_TYPE_INPUT_KEY) in ALL_READ_ONLY_SUBAGENT_TYPES:
|
|
127
|
+
return None
|
|
128
|
+
if tool_name == WORKFLOW_TOOL_NAME:
|
|
129
|
+
return _workflow_script(all_tool_input_fields)
|
|
130
|
+
brief = all_tool_input_fields.get(ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME[tool_name])
|
|
131
|
+
return brief if isinstance(brief, str) else ""
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def scope_settled_line(brief: str) -> str | None:
|
|
71
135
|
"""Return the brief's "Scope settled:" line when it names a reason, else None.
|
|
72
136
|
|
|
73
137
|
::
|
|
@@ -77,12 +141,8 @@ def scope_settled_line(tool_name: str, all_tool_input_fields: dict[str, object])
|
|
|
77
141
|
"Do X.\\nScope settled:" -> None
|
|
78
142
|
|
|
79
143
|
Args:
|
|
80
|
-
|
|
81
|
-
all_tool_input_fields: The spawn input.
|
|
144
|
+
brief: The text the spawn hands its agents.
|
|
82
145
|
"""
|
|
83
|
-
brief = all_tool_input_fields.get(ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME.get(tool_name, ""))
|
|
84
|
-
if not isinstance(brief, str):
|
|
85
|
-
return None
|
|
86
146
|
for each_line in brief.splitlines():
|
|
87
147
|
stripped_line = each_line.strip()
|
|
88
148
|
stated_reason = stripped_line.removeprefix(SCOPE_SETTLED_PREFIX).strip()
|
|
@@ -125,24 +185,35 @@ def _transcript_lines(transcript_path: object) -> list[str] | None:
|
|
|
125
185
|
return None
|
|
126
186
|
|
|
127
187
|
|
|
128
|
-
def
|
|
188
|
+
def _context_output(context: str) -> dict[str, object]:
|
|
129
189
|
return {
|
|
130
190
|
HOOK_SPECIFIC_OUTPUT_KEY: {
|
|
131
191
|
HOOK_EVENT_NAME_KEY: PRE_TOOL_USE_EVENT_NAME,
|
|
132
|
-
|
|
133
|
-
PERMISSION_DECISION_REASON_KEY: reason,
|
|
192
|
+
ADDITIONAL_CONTEXT_KEY: context,
|
|
134
193
|
}
|
|
135
194
|
}
|
|
136
195
|
|
|
137
196
|
|
|
138
|
-
def
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
197
|
+
def _transcript_gaps(
|
|
198
|
+
all_hook_fields: dict[str, object], tool_use_id: str | None
|
|
199
|
+
) -> list[str] | None:
|
|
200
|
+
if all_hook_fields.get(CODEX_TURN_ID_KEY):
|
|
201
|
+
return [CODEX_TRANSCRIPT_REMINDER]
|
|
202
|
+
all_transcript_lines = _transcript_lines(all_hook_fields.get(TRANSCRIPT_PATH_KEY))
|
|
203
|
+
if all_transcript_lines is None:
|
|
204
|
+
return None
|
|
205
|
+
return readiness_gaps(session_steps(all_transcript_lines, tool_use_id))
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _gaps_after_scope_settled(
|
|
209
|
+
all_gaps: list[str], brief: str, tool_name: str, tool_use_id: str | None
|
|
210
|
+
) -> list[str]:
|
|
211
|
+
settled_line = scope_settled_line(brief)
|
|
212
|
+
all_settled_gaps = {MISSING_INTERVIEW_REASON, CODEX_TRANSCRIPT_REMINDER}
|
|
213
|
+
if settled_line is None or not all_settled_gaps.intersection(all_gaps):
|
|
214
|
+
return all_gaps
|
|
215
|
+
_log_decision(tool_name, tool_use_id, OUTCOME_SCOPE_SETTLED, settled_line)
|
|
216
|
+
return [each_gap for each_gap in all_gaps if each_gap not in all_settled_gaps]
|
|
146
217
|
|
|
147
218
|
|
|
148
219
|
def decide_hook_output(all_hook_fields: dict[str, object]) -> dict[str, object] | None:
|
|
@@ -152,34 +223,35 @@ def decide_hook_output(all_hook_fields: dict[str, object]) -> dict[str, object]
|
|
|
152
223
|
all_hook_fields: The parsed PreToolUse payload.
|
|
153
224
|
|
|
154
225
|
Returns:
|
|
155
|
-
None to
|
|
226
|
+
None to stay quiet, else the reminder as additionalContext output.
|
|
156
227
|
"""
|
|
157
|
-
|
|
228
|
+
tool_name = all_hook_fields.get(TOOL_NAME_KEY)
|
|
229
|
+
tool_input = all_hook_fields.get(TOOL_INPUT_KEY)
|
|
230
|
+
if not isinstance(tool_name, str) or not isinstance(tool_input, dict):
|
|
231
|
+
return None
|
|
232
|
+
if all_hook_fields.get(AGENT_ID_KEY):
|
|
233
|
+
return None
|
|
234
|
+
brief = spawn_brief(tool_name, tool_input)
|
|
235
|
+
if brief is None:
|
|
158
236
|
return None
|
|
159
|
-
tool_name = str(all_hook_fields[TOOL_NAME_KEY])
|
|
160
237
|
raw_tool_use_id = all_hook_fields.get(TOOL_USE_ID_KEY)
|
|
161
238
|
tool_use_id = raw_tool_use_id if isinstance(raw_tool_use_id, str) else None
|
|
162
|
-
|
|
163
|
-
if
|
|
239
|
+
all_gaps = _transcript_gaps(all_hook_fields, tool_use_id)
|
|
240
|
+
if all_gaps is None:
|
|
164
241
|
_log_decision(tool_name, tool_use_id, OUTCOME_TRANSCRIPT_UNREADABLE, None)
|
|
165
242
|
return None
|
|
166
|
-
all_gaps =
|
|
167
|
-
settled_line = scope_settled_line(tool_name, all_hook_fields[TOOL_INPUT_KEY])
|
|
168
|
-
if settled_line is not None and MISSING_INTERVIEW_REASON in all_gaps:
|
|
169
|
-
all_gaps.remove(MISSING_INTERVIEW_REASON)
|
|
170
|
-
_log_decision(tool_name, tool_use_id, OUTCOME_SCOPE_SETTLED, settled_line)
|
|
243
|
+
all_gaps = _gaps_after_scope_settled(all_gaps, brief, tool_name, tool_use_id)
|
|
171
244
|
if not all_gaps:
|
|
172
245
|
return None
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
return _deny_output(deny_reason)
|
|
246
|
+
_log_decision(tool_name, tool_use_id, OUTCOME_REMINDED, None)
|
|
247
|
+
return _context_output(REASON_SEPARATOR.join(all_gaps))
|
|
176
248
|
|
|
177
249
|
|
|
178
250
|
def main() -> int:
|
|
179
|
-
"""Read the PreToolUse payload and print
|
|
251
|
+
"""Read the PreToolUse payload and print the reminder when the spawn is not ready.
|
|
180
252
|
|
|
181
253
|
Returns:
|
|
182
|
-
0 in every case;
|
|
254
|
+
0 in every case; the reminder travels in the JSON output.
|
|
183
255
|
"""
|
|
184
256
|
hook_payload = read_hook_input_dictionary_from_stdin()
|
|
185
257
|
if hook_payload is None:
|
|
@@ -8,12 +8,15 @@ import pytest
|
|
|
8
8
|
|
|
9
9
|
import spawn_readiness_hook
|
|
10
10
|
from hooks_constants.spawn_readiness_hook_constants import (
|
|
11
|
+
CODEX_TRANSCRIPT_REMINDER,
|
|
11
12
|
MISSING_INTERVIEW_REASON,
|
|
12
13
|
MISSING_INVESTIGATION_REASON,
|
|
13
14
|
)
|
|
14
15
|
|
|
15
16
|
HOOK_SCRIPT = Path(__file__).resolve().parent / "spawn_readiness_hook.py"
|
|
16
17
|
THREAD_SPAWN_TOOL_NAME = "mcp__hearthbot__start_thread_session"
|
|
18
|
+
CODEX_SPAWN_TOOL_NAME = "multi_agent_v1__spawn_agent"
|
|
19
|
+
DISPATCH_TOOL_NAME = "mcp__github__actions_run_trigger"
|
|
17
20
|
SETTLED_LINE = "Scope settled: the request names the file, the fix, and the test."
|
|
18
21
|
|
|
19
22
|
|
|
@@ -63,6 +66,12 @@ INTERVIEWED_TRANSCRIPT = [_user("Build the hook."), READ_STEP, QUESTION_STEP, RE
|
|
|
63
66
|
def _spawn_input(tool_name: str, brief: str = "Do the task.") -> dict[str, object]:
|
|
64
67
|
if tool_name == THREAD_SPAWN_TOOL_NAME:
|
|
65
68
|
return {"title": "Task", "instructions": brief}
|
|
69
|
+
if tool_name == CODEX_SPAWN_TOOL_NAME:
|
|
70
|
+
return {"message": brief}
|
|
71
|
+
if tool_name == "Workflow":
|
|
72
|
+
return {"script": brief}
|
|
73
|
+
if tool_name == DISPATCH_TOOL_NAME:
|
|
74
|
+
return {"method": "run_workflow", "workflow_id": "build.yml", "inputs": {"prompt": brief}}
|
|
66
75
|
return {"description": "Task", "prompt": brief}
|
|
67
76
|
|
|
68
77
|
|
|
@@ -104,11 +113,11 @@ def _spawn(
|
|
|
104
113
|
)
|
|
105
114
|
|
|
106
115
|
|
|
107
|
-
def
|
|
116
|
+
def _reminder(stdout: str) -> str:
|
|
108
117
|
decision = json.loads(stdout)["hookSpecificOutput"]
|
|
118
|
+
assert set(decision) == {"hookEventName", "additionalContext"}
|
|
109
119
|
assert decision["hookEventName"] == "PreToolUse"
|
|
110
|
-
|
|
111
|
-
return decision["permissionDecisionReason"]
|
|
120
|
+
return decision["additionalContext"]
|
|
112
121
|
|
|
113
122
|
|
|
114
123
|
def _decision_log(tmp_path: Path) -> list[dict[str, object]]:
|
|
@@ -118,30 +127,33 @@ def _decision_log(tmp_path: Path) -> list[dict[str, object]]:
|
|
|
118
127
|
]
|
|
119
128
|
|
|
120
129
|
|
|
121
|
-
|
|
122
|
-
|
|
130
|
+
ALL_TRANSCRIPT_SPAWN_TOOL_NAMES = ["Agent", "Task", THREAD_SPAWN_TOOL_NAME, "Workflow", DISPATCH_TOOL_NAME]
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
@pytest.mark.parametrize("tool_name", ALL_TRANSCRIPT_SPAWN_TOOL_NAMES)
|
|
134
|
+
def test_should_stay_quiet_after_a_read_a_question_and_a_reply(
|
|
123
135
|
tmp_path: Path, tool_name: str
|
|
124
136
|
) -> None:
|
|
125
137
|
assert _spawn(tmp_path, INTERVIEWED_TRANSCRIPT, tool_name) == ""
|
|
126
138
|
|
|
127
139
|
|
|
128
|
-
def
|
|
140
|
+
def test_should_remind_a_spawn_with_no_read_since_the_request(tmp_path: Path) -> None:
|
|
129
141
|
transcript = [_user("Build the hook."), QUESTION_STEP, REPLY_STEP]
|
|
130
|
-
assert
|
|
142
|
+
assert _reminder(_spawn(tmp_path, transcript)) == MISSING_INVESTIGATION_REASON
|
|
131
143
|
|
|
132
144
|
|
|
133
|
-
def
|
|
145
|
+
def test_should_remind_a_spawn_with_no_question_to_the_user(tmp_path: Path) -> None:
|
|
134
146
|
transcript = [_user("Build the hook."), READ_STEP]
|
|
135
|
-
assert
|
|
147
|
+
assert _reminder(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
136
148
|
|
|
137
149
|
|
|
138
|
-
def
|
|
150
|
+
def test_should_remind_a_spawn_whose_question_has_no_reply_yet(tmp_path: Path) -> None:
|
|
139
151
|
transcript = [_user("Build the hook."), READ_STEP, QUESTION_STEP]
|
|
140
|
-
assert
|
|
152
|
+
assert _reminder(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
141
153
|
|
|
142
154
|
|
|
143
155
|
def test_should_name_both_gaps_when_the_session_did_neither(tmp_path: Path) -> None:
|
|
144
|
-
reason =
|
|
156
|
+
reason = _reminder(_spawn(tmp_path, [_user("Build the hook.")]))
|
|
145
157
|
assert reason == f"{MISSING_INVESTIGATION_REASON} {MISSING_INTERVIEW_REASON}"
|
|
146
158
|
|
|
147
159
|
|
|
@@ -150,7 +162,7 @@ def test_should_count_a_plain_chat_question_as_no_interview(
|
|
|
150
162
|
) -> None:
|
|
151
163
|
statement = _tool_use("mcp__hearthbot__post_message", {"text": "Which repo should this land in?"})
|
|
152
164
|
transcript = [_user("Build the hook."), READ_STEP, statement, REPLY_STEP, READ_STEP]
|
|
153
|
-
assert
|
|
165
|
+
assert _reminder(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
154
166
|
|
|
155
167
|
|
|
156
168
|
def test_should_count_an_answered_ask_user_question_as_the_interview(tmp_path: Path) -> None:
|
|
@@ -182,13 +194,13 @@ def test_should_count_an_answered_decision_card_as_the_interview(tmp_path: Path)
|
|
|
182
194
|
def test_should_skip_an_agent_relay_as_the_reply(tmp_path: Path) -> None:
|
|
183
195
|
relay = _queued('<relay from="coordinator"><note>Keep going.</note></relay>')
|
|
184
196
|
transcript = [_user("Build the hook."), READ_STEP, QUESTION_STEP, relay]
|
|
185
|
-
assert
|
|
197
|
+
assert _reminder(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
186
198
|
|
|
187
199
|
|
|
188
200
|
def test_should_require_a_new_interview_for_a_follow_up_request(tmp_path: Path) -> None:
|
|
189
201
|
earlier_spawn = _tool_use("Agent", {"prompt": "Build the hook."})
|
|
190
202
|
transcript = [*INTERVIEWED_TRANSCRIPT, earlier_spawn, _user("Now add a second hook."), READ_STEP]
|
|
191
|
-
assert
|
|
203
|
+
assert _reminder(_spawn(tmp_path, transcript)) == MISSING_INTERVIEW_REASON
|
|
192
204
|
|
|
193
205
|
|
|
194
206
|
def test_should_treat_two_messages_in_a_row_after_a_question_as_one_reply(
|
|
@@ -210,13 +222,13 @@ def test_should_pass_the_interview_on_a_scope_settled_line_and_log_it(tmp_path:
|
|
|
210
222
|
|
|
211
223
|
def test_should_still_require_the_read_with_a_scope_settled_line(tmp_path: Path) -> None:
|
|
212
224
|
brief = f"Fix the typo.\n{SETTLED_LINE}"
|
|
213
|
-
reason =
|
|
225
|
+
reason = _reminder(_spawn(tmp_path, [_user("Fix the typo.")], brief=brief))
|
|
214
226
|
assert reason == MISSING_INVESTIGATION_REASON
|
|
215
227
|
|
|
216
228
|
|
|
217
|
-
def
|
|
229
|
+
def test_should_remind_on_a_scope_settled_line_with_no_reason(tmp_path: Path) -> None:
|
|
218
230
|
transcript = [_user("Fix the typo."), READ_STEP]
|
|
219
|
-
reason =
|
|
231
|
+
reason = _reminder(_spawn(tmp_path, transcript, brief="Fix it.\nScope settled:"))
|
|
220
232
|
assert reason == MISSING_INTERVIEW_REASON
|
|
221
233
|
|
|
222
234
|
|
|
@@ -268,11 +280,91 @@ def test_should_let_the_spawn_run_and_log_when_the_transcript_is_unreadable(
|
|
|
268
280
|
|
|
269
281
|
|
|
270
282
|
def test_should_return_the_scope_settled_line_from_the_brief() -> None:
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
spawn_readiness_hook.scope_settled_line(THREAD_SPAWN_TOOL_NAME, tool_input) == SETTLED_LINE
|
|
283
|
+
brief = spawn_readiness_hook.spawn_brief(
|
|
284
|
+
THREAD_SPAWN_TOOL_NAME, {"instructions": f"Do X.\n {SETTLED_LINE}"}
|
|
274
285
|
)
|
|
286
|
+
assert brief is not None
|
|
287
|
+
assert spawn_readiness_hook.scope_settled_line(brief) == SETTLED_LINE
|
|
275
288
|
|
|
276
289
|
|
|
277
290
|
def test_should_ignore_a_tool_it_does_not_check() -> None:
|
|
278
291
|
assert spawn_readiness_hook.decide_hook_output({"tool_name": "Read", "tool_input": {}}) is None
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
@pytest.mark.parametrize("tool_name", ALL_TRANSCRIPT_SPAWN_TOOL_NAMES)
|
|
295
|
+
def test_should_remind_every_spawn_surface_that_skipped_both_steps(
|
|
296
|
+
tmp_path: Path, tool_name: str
|
|
297
|
+
) -> None:
|
|
298
|
+
reminder = _reminder(_spawn(tmp_path, [_user("Build the hook.")], tool_name))
|
|
299
|
+
assert reminder == f"{MISSING_INVESTIGATION_REASON} {MISSING_INTERVIEW_REASON}"
|
|
300
|
+
[log_record] = _decision_log(tmp_path)
|
|
301
|
+
assert log_record["outcome"] == "reminded"
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def test_should_read_a_workflow_script_from_its_path(tmp_path: Path) -> None:
|
|
305
|
+
script_path = tmp_path / "fan_out.js"
|
|
306
|
+
script_path.write_text(f"const brief = `Fix it.\n{SETTLED_LINE}`\n", encoding="utf-8")
|
|
307
|
+
stdout = _run_hook(
|
|
308
|
+
tmp_path,
|
|
309
|
+
{
|
|
310
|
+
"tool_name": "Workflow",
|
|
311
|
+
"tool_input": {"scriptPath": str(script_path)},
|
|
312
|
+
"transcript_path": str(_write_transcript(tmp_path, [_user("Fix it."), READ_STEP])),
|
|
313
|
+
},
|
|
314
|
+
)
|
|
315
|
+
assert stdout == ""
|
|
316
|
+
[log_record] = _decision_log(tmp_path)
|
|
317
|
+
assert log_record["outcome"] == "scope_settled"
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
@pytest.mark.parametrize(
|
|
321
|
+
"tool_input",
|
|
322
|
+
[
|
|
323
|
+
{"method": "cancel_workflow_run", "run_id": 7},
|
|
324
|
+
{"method": "run_workflow", "workflow_id": "lint.yml", "inputs": {"ref": "main"}},
|
|
325
|
+
{"method": "run_workflow", "workflow_id": "lint.yml"},
|
|
326
|
+
],
|
|
327
|
+
)
|
|
328
|
+
def test_should_ignore_a_workflow_call_that_starts_no_agent(
|
|
329
|
+
tmp_path: Path, tool_input: dict[str, object]
|
|
330
|
+
) -> None:
|
|
331
|
+
stdout = _run_hook(
|
|
332
|
+
tmp_path,
|
|
333
|
+
{
|
|
334
|
+
"tool_name": DISPATCH_TOOL_NAME,
|
|
335
|
+
"tool_input": tool_input,
|
|
336
|
+
"transcript_path": str(_write_transcript(tmp_path, [_user("Build the hook.")])),
|
|
337
|
+
},
|
|
338
|
+
)
|
|
339
|
+
assert stdout == ""
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
@pytest.mark.parametrize("tool_name", [CODEX_SPAWN_TOOL_NAME, "Agent"])
|
|
343
|
+
def test_should_give_a_codex_spawn_the_codex_reminder_without_reading_the_transcript(
|
|
344
|
+
tmp_path: Path, tool_name: str
|
|
345
|
+
) -> None:
|
|
346
|
+
stdout = _run_hook(
|
|
347
|
+
tmp_path,
|
|
348
|
+
{
|
|
349
|
+
"tool_name": tool_name,
|
|
350
|
+
"turn_id": "turn_1",
|
|
351
|
+
"tool_input": _spawn_input(tool_name),
|
|
352
|
+
"transcript_path": str(_write_transcript(tmp_path, INTERVIEWED_TRANSCRIPT)),
|
|
353
|
+
},
|
|
354
|
+
)
|
|
355
|
+
assert _reminder(stdout) == CODEX_TRANSCRIPT_REMINDER
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def test_should_quiet_a_codex_spawn_whose_brief_settles_scope(tmp_path: Path) -> None:
|
|
359
|
+
stdout = _run_hook(
|
|
360
|
+
tmp_path,
|
|
361
|
+
{
|
|
362
|
+
"tool_name": CODEX_SPAWN_TOOL_NAME,
|
|
363
|
+
"turn_id": "turn_1",
|
|
364
|
+
"tool_input": _spawn_input(CODEX_SPAWN_TOOL_NAME, f"Fix it.\n{SETTLED_LINE}"),
|
|
365
|
+
"transcript_path": None,
|
|
366
|
+
},
|
|
367
|
+
)
|
|
368
|
+
assert stdout == ""
|
|
369
|
+
[log_record] = _decision_log(tmp_path)
|
|
370
|
+
assert log_record["outcome"] == "scope_settled"
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import json
|
|
2
2
|
|
|
3
|
+
from hooks_constants.spawn_readiness_hook_constants import MISSING_INVESTIGATION_REASON
|
|
3
4
|
from spawn_readiness_steps import SessionStep, readiness_gaps, session_steps
|
|
4
5
|
|
|
5
6
|
|
|
@@ -81,7 +82,4 @@ def test_should_start_the_span_at_the_request_before_an_answered_question() -> N
|
|
|
81
82
|
SessionStep.QUESTION,
|
|
82
83
|
SessionStep.USER_MESSAGE,
|
|
83
84
|
]
|
|
84
|
-
assert readiness_gaps(all_steps) == [
|
|
85
|
-
"Investigate the request before this spawn. Read the files, threads, or sources it names, "
|
|
86
|
-
"so the brief and the agent count fit the task. Run the reads in a message before the spawn."
|
|
87
|
-
]
|
|
85
|
+
assert readiness_gaps(all_steps) == [MISSING_INVESTIGATION_REASON]
|
|
@@ -29,8 +29,6 @@ if _hooks_dir not in sys.path:
|
|
|
29
29
|
|
|
30
30
|
from hooks_constants.skill_loaded_reminder_constants import (
|
|
31
31
|
ALL_SELF_LOADING_SUBAGENT_TYPES,
|
|
32
|
-
ASSISTANT_ENTRY_TYPE,
|
|
33
|
-
COMPACT_BOUNDARY_SUBTYPE,
|
|
34
32
|
COMPACTION_REMINDER,
|
|
35
33
|
COMPACTION_SOURCE,
|
|
36
34
|
NOT_LOADED_REMINDER,
|
|
@@ -38,18 +36,16 @@ from hooks_constants.skill_loaded_reminder_constants import (
|
|
|
38
36
|
PRE_TOOL_USE_EVENT_NAME,
|
|
39
37
|
PROMPT_SEPARATOR,
|
|
40
38
|
SESSION_START_EVENT_NAME,
|
|
41
|
-
SKILL_TOOL_NAME,
|
|
42
39
|
ALL_SLASH_COMMAND_MARKERS,
|
|
43
40
|
SUBAGENT_START_EVENT_NAME,
|
|
44
41
|
ALL_SPAWN_PROMPT_FIELDS_AND_PREFIXES_BY_TOOL_NAME,
|
|
45
|
-
TOOL_USE_BLOCK_TYPE,
|
|
46
|
-
USER_ENTRY_TYPE,
|
|
47
42
|
USER_PROMPT_SUBMIT_EVENT_NAME,
|
|
48
43
|
WORKFLOW_SUBAGENT_TYPE,
|
|
49
44
|
)
|
|
50
45
|
from hooks_constants.pre_tool_use_allow_output import write_pre_tool_use_allow_to_stdout
|
|
51
46
|
from hooks_constants.pre_tool_use_stdin import read_hook_input_dictionary_from_stdin
|
|
52
47
|
from hooks_constants.setup_project_paths_constants import DECODE_ERRORS_POLICY, UTF8_ENCODING
|
|
48
|
+
from transcript_skill_scan import skill_invocation_status
|
|
53
49
|
|
|
54
50
|
|
|
55
51
|
def subagent_input_with_poteto_mode(
|
|
@@ -81,44 +77,6 @@ def subagent_input_with_poteto_mode(
|
|
|
81
77
|
return {**all_tool_input_fields, field_name: invocation_prefix + PROMPT_SEPARATOR + prompt}
|
|
82
78
|
|
|
83
79
|
|
|
84
|
-
def _invokes_poteto_mode(all_entry_fields: dict[str, object]) -> bool:
|
|
85
|
-
message = all_entry_fields.get("message")
|
|
86
|
-
if not isinstance(message, dict):
|
|
87
|
-
return False
|
|
88
|
-
content_blocks = message.get("content")
|
|
89
|
-
if all_entry_fields.get("type") == USER_ENTRY_TYPE and isinstance(content_blocks, str):
|
|
90
|
-
return any(each_marker in content_blocks for each_marker in ALL_SLASH_COMMAND_MARKERS)
|
|
91
|
-
if all_entry_fields.get("type") != ASSISTANT_ENTRY_TYPE or not isinstance(content_blocks, list):
|
|
92
|
-
return False
|
|
93
|
-
return any(
|
|
94
|
-
isinstance(each_block, dict)
|
|
95
|
-
and each_block.get("type") == TOOL_USE_BLOCK_TYPE
|
|
96
|
-
and each_block.get("name") == SKILL_TOOL_NAME
|
|
97
|
-
and isinstance(each_block.get("input"), dict)
|
|
98
|
-
and each_block["input"].get("skill") in ALL_POTETO_MODE_SKILL_NAMES
|
|
99
|
-
for each_block in content_blocks
|
|
100
|
-
)
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
def _marker_entry(transcript_line: str) -> dict[str, object] | None:
|
|
104
|
-
if (
|
|
105
|
-
COMPACT_BOUNDARY_SUBTYPE not in transcript_line
|
|
106
|
-
and ALL_POTETO_MODE_SKILL_NAMES[0] not in transcript_line
|
|
107
|
-
):
|
|
108
|
-
return None
|
|
109
|
-
try:
|
|
110
|
-
parsed_entry = json.loads(transcript_line)
|
|
111
|
-
except json.JSONDecodeError:
|
|
112
|
-
return None
|
|
113
|
-
return parsed_entry if isinstance(parsed_entry, dict) else None
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
def _is_loaded_after(all_entry_fields: dict[str, object], was_loaded: bool) -> bool:
|
|
117
|
-
if all_entry_fields.get("subtype") == COMPACT_BOUNDARY_SUBTYPE:
|
|
118
|
-
return False
|
|
119
|
-
return was_loaded or _invokes_poteto_mode(all_entry_fields)
|
|
120
|
-
|
|
121
|
-
|
|
122
80
|
def poteto_mode_status(all_transcript_lines: Iterable[str]) -> tuple[bool, bool]:
|
|
123
81
|
"""Report whether the skill was invoked and whether it remains loaded.
|
|
124
82
|
|
|
@@ -139,14 +97,9 @@ def poteto_mode_status(all_transcript_lines: Iterable[str]) -> tuple[bool, bool]
|
|
|
139
97
|
Returns:
|
|
140
98
|
Whether the skill was invoked at any point and whether it is loaded now.
|
|
141
99
|
"""
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
all_entry_fields = _marker_entry(each_line)
|
|
146
|
-
if all_entry_fields is not None:
|
|
147
|
-
is_loaded = _is_loaded_after(all_entry_fields, is_loaded)
|
|
148
|
-
was_invoked = was_invoked or is_loaded
|
|
149
|
-
return was_invoked, is_loaded
|
|
100
|
+
return skill_invocation_status(
|
|
101
|
+
all_transcript_lines, ALL_POTETO_MODE_SKILL_NAMES, ALL_SLASH_COMMAND_MARKERS
|
|
102
|
+
)
|
|
150
103
|
|
|
151
104
|
|
|
152
105
|
def is_poteto_mode_loaded(all_transcript_lines: Iterable[str]) -> bool:
|
|
@@ -200,6 +200,10 @@ class TestIsPotetoModeLoaded:
|
|
|
200
200
|
prefixed_command_line = TYPED_COMMAND_LINE.replace("/poteto-mode", "/pstack:poteto-mode")
|
|
201
201
|
assert reminder.is_poteto_mode_loaded([prefixed_command_line])
|
|
202
202
|
|
|
203
|
+
def test_a_different_skill_call_does_not_load_it(self) -> None:
|
|
204
|
+
other_skill_call_line = SKILL_CALL_LINE.replace("poteto-mode", "pr-lifecycle")
|
|
205
|
+
assert not reminder.is_poteto_mode_loaded([other_skill_call_line])
|
|
206
|
+
|
|
203
207
|
def test_a_compaction_after_the_skill_call_drops_it(self) -> None:
|
|
204
208
|
assert not reminder.is_poteto_mode_loaded(
|
|
205
209
|
[SKILL_CALL_LINE, COMPACT_BOUNDARY_LINE, READ_CALL_LINE]
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Tests for skill invocation state across transcript compactions."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import sys
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
HOOKS_DIRECTORY = Path(__file__).resolve().parent
|
|
8
|
+
if str(HOOKS_DIRECTORY) not in sys.path:
|
|
9
|
+
sys.path.insert(0, str(HOOKS_DIRECTORY))
|
|
10
|
+
|
|
11
|
+
from transcript_skill_scan import is_skill_loaded_after_last_compaction, skill_invocation_status
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _skill_entry(name: str) -> str:
|
|
15
|
+
return json.dumps(
|
|
16
|
+
{"type": "assistant", "message": {"content": [{"type": "tool_use", "name": "Skill", "input": {"skill": name}}]}}
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def test_matching_skill_after_compaction_loads() -> None:
|
|
21
|
+
lines = [
|
|
22
|
+
_skill_entry("pr-lifecycle"),
|
|
23
|
+
json.dumps({"type": "system", "subtype": "compact_boundary"}),
|
|
24
|
+
_skill_entry("plugin:pr-lifecycle"),
|
|
25
|
+
]
|
|
26
|
+
assert is_skill_loaded_after_last_compaction(
|
|
27
|
+
lines, ("pr-lifecycle",), ("<command-name>/pr-lifecycle</command-name>",)
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def test_compaction_drops_a_loaded_skill() -> None:
|
|
32
|
+
lines = [_skill_entry("pr-lifecycle"), json.dumps({"subtype": "compact_boundary"})]
|
|
33
|
+
assert not is_skill_loaded_after_last_compaction(lines, ("pr-lifecycle",), ())
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_user_command_loads_and_quoted_tool_result_does_not() -> None:
|
|
37
|
+
marker = "<command-name>/pr-lifecycle</command-name>"
|
|
38
|
+
user_entry = json.dumps({"type": "user", "message": {"content": marker}})
|
|
39
|
+
quoted_result = json.dumps(
|
|
40
|
+
{"type": "user", "message": {"content": [{"type": "tool_result", "content": marker}]}}
|
|
41
|
+
)
|
|
42
|
+
assert is_skill_loaded_after_last_compaction([user_entry], ("pr-lifecycle",), (marker,))
|
|
43
|
+
assert not is_skill_loaded_after_last_compaction([quoted_result], ("pr-lifecycle",), (marker,))
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def test_malformed_entries_do_not_load() -> None:
|
|
47
|
+
assert not is_skill_loaded_after_last_compaction(
|
|
48
|
+
["pr-lifecycle {", json.dumps({"type": "assistant", "message": {"content": "pr-lifecycle"}})],
|
|
49
|
+
("pr-lifecycle",),
|
|
50
|
+
(),
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def test_invocation_status_separates_ever_invoked_from_loaded_now() -> None:
|
|
55
|
+
compaction = json.dumps({"subtype": "compact_boundary"})
|
|
56
|
+
assert skill_invocation_status([_skill_entry("pr-lifecycle")], ("pr-lifecycle",), ()) == (True, True)
|
|
57
|
+
assert skill_invocation_status([_skill_entry("pr-lifecycle"), compaction], ("pr-lifecycle",), ()) == (True, False)
|
|
58
|
+
assert skill_invocation_status([compaction, _skill_entry("other")], ("pr-lifecycle",), ()) == (False, False)
|