claude-dev-env 8.36.4 → 8.37.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/rule-guides/asd-ste100-language.md +42 -0
- package/docs/rule-guides/cleanup-temp-files.md +35 -0
- package/docs/rule-guides/correction-lens.md +112 -0
- package/docs/rule-guides/destructive-commands.md +49 -0
- package/docs/rule-guides/explore-thoroughly.md +27 -0
- package/docs/rule-guides/filesystem-search.md +53 -0
- package/docs/rule-guides/memory-stores-durable-facts.md +59 -0
- package/docs/rule-guides/no-contrast-framing.md +70 -0
- package/docs/rule-guides/research-mode.md +31 -0
- package/docs/rule-guides/verify-before-asking.md +54 -0
- package/docs/rule-guides/verify-runtime-state.md +44 -0
- package/hooks/hooks.json +10 -0
- package/hooks/hooks_constants/spawn_readiness_hook_constants.py +111 -0
- package/hooks/routing/spawn_readiness_hook.py +195 -0
- package/hooks/routing/spawn_readiness_steps.py +266 -0
- package/hooks/routing/test_spawn_readiness_hook.py +278 -0
- package/hooks/routing/test_spawn_readiness_steps.py +87 -0
- package/package.json +1 -1
- package/rules/asd-ste100-language.md +5 -36
- package/rules/cleanup-temp-files.md +5 -29
- package/rules/correction-lens.md +13 -102
- package/rules/destructive-commands.md +5 -41
- package/rules/explore-thoroughly.md +5 -21
- package/rules/filesystem-search.md +5 -47
- package/rules/memory-stores-durable-facts.md +5 -53
- package/rules/no-contrast-framing.md +5 -64
- package/rules/research-mode.md +5 -25
- package/rules/verify-before-asking.md +5 -48
- package/rules/verify-runtime-state.md +5 -38
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""Constants for the spawn readiness PreToolUse hook.
|
|
2
|
+
|
|
3
|
+
Groups: the spawn tools it checks and their brief fields, the transcript entry
|
|
4
|
+
shapes it reads, the tool names that count as a read or a question, the
|
|
5
|
+
scope-settled line, and the deny messages and log fields.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
TOOL_NAME_KEY = "tool_name"
|
|
11
|
+
TOOL_INPUT_KEY = "tool_input"
|
|
12
|
+
TOOL_USE_ID_KEY = "tool_use_id"
|
|
13
|
+
TRANSCRIPT_PATH_KEY = "transcript_path"
|
|
14
|
+
AGENT_ID_KEY = "agent_id"
|
|
15
|
+
SUBAGENT_TYPE_INPUT_KEY = "subagent_type"
|
|
16
|
+
|
|
17
|
+
THREAD_SPAWN_TOOL_NAME = "mcp__hearthbot__start_thread_session"
|
|
18
|
+
ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME = {
|
|
19
|
+
"Agent": "prompt",
|
|
20
|
+
"Task": "prompt",
|
|
21
|
+
THREAD_SPAWN_TOOL_NAME: "instructions",
|
|
22
|
+
}
|
|
23
|
+
ALL_READ_ONLY_SUBAGENT_TYPES = frozenset({"Explore", "Plan", "claude-code-guide"})
|
|
24
|
+
|
|
25
|
+
ENTRY_TYPE_KEY = "type"
|
|
26
|
+
USER_ENTRY_TYPE = "user"
|
|
27
|
+
ASSISTANT_ENTRY_TYPE = "assistant"
|
|
28
|
+
ATTACHMENT_ENTRY_TYPE = "attachment"
|
|
29
|
+
QUEUED_COMMAND_ATTACHMENT_TYPE = "queued_command"
|
|
30
|
+
ATTACHMENT_KEY = "attachment"
|
|
31
|
+
QUEUED_PROMPT_KEY = "prompt"
|
|
32
|
+
MESSAGE_KEY = "message"
|
|
33
|
+
CONTENT_KEY = "content"
|
|
34
|
+
IS_META_KEY = "isMeta"
|
|
35
|
+
IS_COMPACT_SUMMARY_KEY = "isCompactSummary"
|
|
36
|
+
BLOCK_TYPE_KEY = "type"
|
|
37
|
+
TEXT_BLOCK_TYPE = "text"
|
|
38
|
+
TEXT_KEY = "text"
|
|
39
|
+
TOOL_USE_BLOCK_TYPE = "tool_use"
|
|
40
|
+
TOOL_RESULT_BLOCK_TYPE = "tool_result"
|
|
41
|
+
BLOCK_ID_KEY = "id"
|
|
42
|
+
BLOCK_NAME_KEY = "name"
|
|
43
|
+
BLOCK_INPUT_KEY = "input"
|
|
44
|
+
RESULT_TOOL_USE_ID_KEY = "tool_use_id"
|
|
45
|
+
RESULT_IS_ERROR_KEY = "is_error"
|
|
46
|
+
|
|
47
|
+
ALL_HARNESS_ENVELOPE_MARKERS = (
|
|
48
|
+
"<wake",
|
|
49
|
+
"<relay",
|
|
50
|
+
"<task-notification",
|
|
51
|
+
"<cross-session-message",
|
|
52
|
+
"<system-note",
|
|
53
|
+
"<local-command",
|
|
54
|
+
)
|
|
55
|
+
HUMAN_SENDER_MARKER = 'from="human"'
|
|
56
|
+
|
|
57
|
+
ALL_READ_TOOL_NAMES = frozenset(
|
|
58
|
+
{
|
|
59
|
+
"Read",
|
|
60
|
+
"Grep",
|
|
61
|
+
"Glob",
|
|
62
|
+
"Bash",
|
|
63
|
+
"PowerShell",
|
|
64
|
+
"LS",
|
|
65
|
+
"NotebookRead",
|
|
66
|
+
"WebFetch",
|
|
67
|
+
"WebSearch",
|
|
68
|
+
}
|
|
69
|
+
)
|
|
70
|
+
MCP_TOOL_NAME_PREFIX = "mcp__"
|
|
71
|
+
MCP_TOOL_NAME_SEPARATOR = "__"
|
|
72
|
+
ALL_MCP_READ_VERB_PREFIXES = ("fetch", "read", "get", "list", "search", "query")
|
|
73
|
+
ALL_INTERACTIVE_QUESTION_TOOL_NAMES = frozenset(
|
|
74
|
+
{"AskUserQuestion", "request_user_input", "request_user_input_async"}
|
|
75
|
+
)
|
|
76
|
+
ALL_INTERACTIVE_MCP_ACTIONS = frozenset({"ask_decision", "post_widget"})
|
|
77
|
+
|
|
78
|
+
SCOPE_SETTLED_PREFIX = "Scope settled:"
|
|
79
|
+
|
|
80
|
+
PRE_TOOL_USE_EVENT_NAME = "PreToolUse"
|
|
81
|
+
HOOK_SPECIFIC_OUTPUT_KEY = "hookSpecificOutput"
|
|
82
|
+
HOOK_EVENT_NAME_KEY = "hookEventName"
|
|
83
|
+
PERMISSION_DECISION_KEY = "permissionDecision"
|
|
84
|
+
PERMISSION_DECISION_REASON_KEY = "permissionDecisionReason"
|
|
85
|
+
PERMISSION_DENY = "deny"
|
|
86
|
+
MISSING_INVESTIGATION_REASON = (
|
|
87
|
+
"Investigate the request before this spawn. Read the files, threads, or "
|
|
88
|
+
"sources it names, so the brief and the agent count fit the task. Run the "
|
|
89
|
+
"reads in a message before the spawn."
|
|
90
|
+
)
|
|
91
|
+
MISSING_INTERVIEW_REASON = (
|
|
92
|
+
"Interview the user before this spawn. Ask your scope, requirements, and "
|
|
93
|
+
"goals questions through AskUserQuestion, a decision card, or an "
|
|
94
|
+
"interactive widget, and spawn after the answer. When the request "
|
|
95
|
+
"already settles scope, add a brief line that starts with "
|
|
96
|
+
f'"{SCOPE_SETTLED_PREFIX}" and names the reason.'
|
|
97
|
+
)
|
|
98
|
+
REASON_SEPARATOR = " "
|
|
99
|
+
|
|
100
|
+
TRANSCRIPT_ENCODING = "utf-8"
|
|
101
|
+
TRANSCRIPT_DECODE_ERRORS = "replace"
|
|
102
|
+
DECISION_LOG_RELATIVE_PATH = ".claude/logs/spawn-readiness.jsonl"
|
|
103
|
+
LOG_APPEND_MODE = "a"
|
|
104
|
+
LOG_LINE_END = "\n"
|
|
105
|
+
LOG_TIMESTAMP_KEY = "timestamp"
|
|
106
|
+
LOG_TOOL_NAME_KEY = "tool_name"
|
|
107
|
+
LOG_TOOL_USE_ID_KEY = "tool_use_id"
|
|
108
|
+
LOG_OUTCOME_KEY = "outcome"
|
|
109
|
+
LOG_SCOPE_SETTLED_LINE_KEY = "scope_settled_line"
|
|
110
|
+
OUTCOME_SCOPE_SETTLED = "scope_settled"
|
|
111
|
+
OUTCOME_TRANSCRIPT_UNREADABLE = "transcript_unreadable"
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""PreToolUse hook: hold an agent spawn until the session has read and asked.
|
|
3
|
+
|
|
4
|
+
Registered on ``Agent``, ``Task``, and ``mcp__hearthbot__start_thread_session``.
|
|
5
|
+
It reads the session transcript and checks two things since the user's request:
|
|
6
|
+
|
|
7
|
+
::
|
|
8
|
+
|
|
9
|
+
no read step after the request -> deny: investigate first
|
|
10
|
+
no interactive question, then an answer -> deny: interview first
|
|
11
|
+
brief line "Scope settled: <reason>" -> interview check passes, logged
|
|
12
|
+
both found -> no output; the call runs
|
|
13
|
+
|
|
14
|
+
``spawn_readiness_steps`` reads the transcript and finds the request. Four
|
|
15
|
+
cases pass without a check: a call from inside a subagent, a read-only
|
|
16
|
+
subagent type, a tool input that is not an object, and a transcript the hook
|
|
17
|
+
cannot read. The last one is logged.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import datetime
|
|
23
|
+
import json
|
|
24
|
+
import sys
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
hooks_root_directory = str(Path(__file__).resolve().parent.parent)
|
|
28
|
+
if hooks_root_directory not in sys.path:
|
|
29
|
+
sys.path.insert(0, hooks_root_directory)
|
|
30
|
+
routing_directory = str(Path(__file__).resolve().parent)
|
|
31
|
+
if routing_directory not in sys.path:
|
|
32
|
+
sys.path.insert(0, routing_directory)
|
|
33
|
+
|
|
34
|
+
from hooks_constants.hook_block_logger import log_hook_block
|
|
35
|
+
from hooks_constants.pre_tool_use_stdin import read_hook_input_dictionary_from_stdin
|
|
36
|
+
from hooks_constants.spawn_readiness_hook_constants import (
|
|
37
|
+
AGENT_ID_KEY,
|
|
38
|
+
ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME,
|
|
39
|
+
ALL_READ_ONLY_SUBAGENT_TYPES,
|
|
40
|
+
DECISION_LOG_RELATIVE_PATH,
|
|
41
|
+
HOOK_EVENT_NAME_KEY,
|
|
42
|
+
HOOK_SPECIFIC_OUTPUT_KEY,
|
|
43
|
+
LOG_APPEND_MODE,
|
|
44
|
+
LOG_LINE_END,
|
|
45
|
+
LOG_OUTCOME_KEY,
|
|
46
|
+
LOG_SCOPE_SETTLED_LINE_KEY,
|
|
47
|
+
LOG_TIMESTAMP_KEY,
|
|
48
|
+
LOG_TOOL_NAME_KEY,
|
|
49
|
+
LOG_TOOL_USE_ID_KEY,
|
|
50
|
+
MISSING_INTERVIEW_REASON,
|
|
51
|
+
OUTCOME_SCOPE_SETTLED,
|
|
52
|
+
OUTCOME_TRANSCRIPT_UNREADABLE,
|
|
53
|
+
PERMISSION_DECISION_KEY,
|
|
54
|
+
PERMISSION_DECISION_REASON_KEY,
|
|
55
|
+
PERMISSION_DENY,
|
|
56
|
+
PRE_TOOL_USE_EVENT_NAME,
|
|
57
|
+
REASON_SEPARATOR,
|
|
58
|
+
SCOPE_SETTLED_PREFIX,
|
|
59
|
+
SUBAGENT_TYPE_INPUT_KEY,
|
|
60
|
+
TOOL_INPUT_KEY,
|
|
61
|
+
TOOL_NAME_KEY,
|
|
62
|
+
TOOL_USE_ID_KEY,
|
|
63
|
+
TRANSCRIPT_DECODE_ERRORS,
|
|
64
|
+
TRANSCRIPT_ENCODING,
|
|
65
|
+
TRANSCRIPT_PATH_KEY,
|
|
66
|
+
)
|
|
67
|
+
from spawn_readiness_steps import readiness_gaps, session_steps
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def scope_settled_line(tool_name: str, all_tool_input_fields: dict[str, object]) -> str | None:
|
|
71
|
+
"""Return the brief's "Scope settled:" line when it names a reason, else None.
|
|
72
|
+
|
|
73
|
+
::
|
|
74
|
+
|
|
75
|
+
"Do X.\\nScope settled: the request names the file and the fix."
|
|
76
|
+
-> "Scope settled: the request names the file and the fix."
|
|
77
|
+
"Do X.\\nScope settled:" -> None
|
|
78
|
+
|
|
79
|
+
Args:
|
|
80
|
+
tool_name: The spawn tool, which decides the brief field.
|
|
81
|
+
all_tool_input_fields: The spawn input.
|
|
82
|
+
"""
|
|
83
|
+
brief = all_tool_input_fields.get(ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME.get(tool_name, ""))
|
|
84
|
+
if not isinstance(brief, str):
|
|
85
|
+
return None
|
|
86
|
+
for each_line in brief.splitlines():
|
|
87
|
+
stripped_line = each_line.strip()
|
|
88
|
+
stated_reason = stripped_line.removeprefix(SCOPE_SETTLED_PREFIX).strip()
|
|
89
|
+
if stripped_line.startswith(SCOPE_SETTLED_PREFIX) and stated_reason:
|
|
90
|
+
return stripped_line
|
|
91
|
+
return None
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _log_decision(
|
|
95
|
+
tool_name: str, tool_use_id: str | None, outcome: str, settled_line: str | None
|
|
96
|
+
) -> None:
|
|
97
|
+
try:
|
|
98
|
+
log_path = Path.home() / DECISION_LOG_RELATIVE_PATH
|
|
99
|
+
except RuntimeError:
|
|
100
|
+
return
|
|
101
|
+
log_record: dict[str, object] = {
|
|
102
|
+
LOG_TIMESTAMP_KEY: datetime.datetime.now().isoformat(),
|
|
103
|
+
LOG_TOOL_NAME_KEY: tool_name,
|
|
104
|
+
LOG_TOOL_USE_ID_KEY: tool_use_id,
|
|
105
|
+
LOG_OUTCOME_KEY: outcome,
|
|
106
|
+
}
|
|
107
|
+
if settled_line is not None:
|
|
108
|
+
log_record[LOG_SCOPE_SETTLED_LINE_KEY] = settled_line
|
|
109
|
+
try:
|
|
110
|
+
log_path.parent.mkdir(parents=True, exist_ok=True)
|
|
111
|
+
with log_path.open(LOG_APPEND_MODE, encoding=TRANSCRIPT_ENCODING) as log_file:
|
|
112
|
+
log_file.write(json.dumps(log_record) + LOG_LINE_END)
|
|
113
|
+
except OSError:
|
|
114
|
+
pass
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _transcript_lines(transcript_path: object) -> list[str] | None:
|
|
118
|
+
if not isinstance(transcript_path, str) or not transcript_path:
|
|
119
|
+
return None
|
|
120
|
+
try:
|
|
121
|
+
return Path(transcript_path).read_text(
|
|
122
|
+
encoding=TRANSCRIPT_ENCODING, errors=TRANSCRIPT_DECODE_ERRORS
|
|
123
|
+
).splitlines()
|
|
124
|
+
except OSError:
|
|
125
|
+
return None
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _deny_output(reason: str) -> dict[str, object]:
|
|
129
|
+
return {
|
|
130
|
+
HOOK_SPECIFIC_OUTPUT_KEY: {
|
|
131
|
+
HOOK_EVENT_NAME_KEY: PRE_TOOL_USE_EVENT_NAME,
|
|
132
|
+
PERMISSION_DECISION_KEY: PERMISSION_DENY,
|
|
133
|
+
PERMISSION_DECISION_REASON_KEY: reason,
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _is_checked_spawn(all_hook_fields: dict[str, object]) -> bool:
|
|
139
|
+
tool_input = all_hook_fields.get(TOOL_INPUT_KEY)
|
|
140
|
+
return (
|
|
141
|
+
all_hook_fields.get(TOOL_NAME_KEY) in ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME
|
|
142
|
+
and isinstance(tool_input, dict)
|
|
143
|
+
and not all_hook_fields.get(AGENT_ID_KEY)
|
|
144
|
+
and tool_input.get(SUBAGENT_TYPE_INPUT_KEY) not in ALL_READ_ONLY_SUBAGENT_TYPES
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def decide_hook_output(all_hook_fields: dict[str, object]) -> dict[str, object] | None:
|
|
149
|
+
"""Choose the hook's output for one spawn call.
|
|
150
|
+
|
|
151
|
+
Args:
|
|
152
|
+
all_hook_fields: The parsed PreToolUse payload.
|
|
153
|
+
|
|
154
|
+
Returns:
|
|
155
|
+
None to let the call run, else the deny JSON output.
|
|
156
|
+
"""
|
|
157
|
+
if not _is_checked_spawn(all_hook_fields):
|
|
158
|
+
return None
|
|
159
|
+
tool_name = str(all_hook_fields[TOOL_NAME_KEY])
|
|
160
|
+
raw_tool_use_id = all_hook_fields.get(TOOL_USE_ID_KEY)
|
|
161
|
+
tool_use_id = raw_tool_use_id if isinstance(raw_tool_use_id, str) else None
|
|
162
|
+
all_transcript_lines = _transcript_lines(all_hook_fields.get(TRANSCRIPT_PATH_KEY))
|
|
163
|
+
if all_transcript_lines is None:
|
|
164
|
+
_log_decision(tool_name, tool_use_id, OUTCOME_TRANSCRIPT_UNREADABLE, None)
|
|
165
|
+
return None
|
|
166
|
+
all_gaps = readiness_gaps(session_steps(all_transcript_lines, tool_use_id))
|
|
167
|
+
settled_line = scope_settled_line(tool_name, all_hook_fields[TOOL_INPUT_KEY])
|
|
168
|
+
if settled_line is not None and MISSING_INTERVIEW_REASON in all_gaps:
|
|
169
|
+
all_gaps.remove(MISSING_INTERVIEW_REASON)
|
|
170
|
+
_log_decision(tool_name, tool_use_id, OUTCOME_SCOPE_SETTLED, settled_line)
|
|
171
|
+
if not all_gaps:
|
|
172
|
+
return None
|
|
173
|
+
deny_reason = REASON_SEPARATOR.join(all_gaps)
|
|
174
|
+
log_hook_block(Path(__file__).name, PRE_TOOL_USE_EVENT_NAME, deny_reason, tool_name=tool_name)
|
|
175
|
+
return _deny_output(deny_reason)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def main() -> int:
|
|
179
|
+
"""Read the PreToolUse payload and print a deny when the spawn is not ready.
|
|
180
|
+
|
|
181
|
+
Returns:
|
|
182
|
+
0 in every case; a deny travels in the JSON output.
|
|
183
|
+
"""
|
|
184
|
+
hook_payload = read_hook_input_dictionary_from_stdin()
|
|
185
|
+
if hook_payload is None:
|
|
186
|
+
return 0
|
|
187
|
+
hook_output = decide_hook_output(hook_payload)
|
|
188
|
+
if hook_output is not None:
|
|
189
|
+
sys.stdout.write(json.dumps(hook_output))
|
|
190
|
+
sys.stdout.flush()
|
|
191
|
+
return 0
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
if __name__ == "__main__":
|
|
195
|
+
sys.exit(main())
|
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
"""Read a session transcript into the steps the spawn readiness hook checks.
|
|
2
|
+
|
|
3
|
+
::
|
|
4
|
+
|
|
5
|
+
user "build X" ... Bash ... AskUserQuestion "Which layout?" ... answer "grid"
|
|
6
|
+
-> [USER_MESSAGE, READ, QUESTION, USER_MESSAGE]
|
|
7
|
+
-> readiness_gaps(...) == []
|
|
8
|
+
|
|
9
|
+
A read is a file read, a search, a shell command, a fetch, an MCP read, or a
|
|
10
|
+
read-only subagent. A question is an interactive prompt: an
|
|
11
|
+
``AskUserQuestion`` or ``request_user_input`` call, whose answer is the
|
|
12
|
+
reply, or an ``ask_decision`` card or ``post_widget`` widget, answered by
|
|
13
|
+
the next user message. A plain chat message is no question. A user message is typed text or a human wake; a harness
|
|
14
|
+
envelope with no human sender, such as an agent relay or a task notice, is
|
|
15
|
+
left out.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import enum
|
|
21
|
+
import json
|
|
22
|
+
import sys
|
|
23
|
+
from collections.abc import Iterable, Iterator
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
|
|
26
|
+
hooks_root_directory = str(Path(__file__).resolve().parent.parent)
|
|
27
|
+
if hooks_root_directory not in sys.path:
|
|
28
|
+
sys.path.insert(0, hooks_root_directory)
|
|
29
|
+
|
|
30
|
+
from hooks_constants.spawn_readiness_hook_constants import (
|
|
31
|
+
ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME,
|
|
32
|
+
ALL_HARNESS_ENVELOPE_MARKERS,
|
|
33
|
+
ALL_MCP_READ_VERB_PREFIXES,
|
|
34
|
+
ALL_READ_ONLY_SUBAGENT_TYPES,
|
|
35
|
+
ALL_READ_TOOL_NAMES,
|
|
36
|
+
ALL_INTERACTIVE_MCP_ACTIONS,
|
|
37
|
+
ALL_INTERACTIVE_QUESTION_TOOL_NAMES,
|
|
38
|
+
ASSISTANT_ENTRY_TYPE,
|
|
39
|
+
ATTACHMENT_ENTRY_TYPE,
|
|
40
|
+
ATTACHMENT_KEY,
|
|
41
|
+
BLOCK_ID_KEY,
|
|
42
|
+
BLOCK_INPUT_KEY,
|
|
43
|
+
BLOCK_NAME_KEY,
|
|
44
|
+
BLOCK_TYPE_KEY,
|
|
45
|
+
CONTENT_KEY,
|
|
46
|
+
ENTRY_TYPE_KEY,
|
|
47
|
+
HUMAN_SENDER_MARKER,
|
|
48
|
+
IS_COMPACT_SUMMARY_KEY,
|
|
49
|
+
IS_META_KEY,
|
|
50
|
+
MCP_TOOL_NAME_PREFIX,
|
|
51
|
+
MCP_TOOL_NAME_SEPARATOR,
|
|
52
|
+
MESSAGE_KEY,
|
|
53
|
+
MISSING_INTERVIEW_REASON,
|
|
54
|
+
MISSING_INVESTIGATION_REASON,
|
|
55
|
+
QUEUED_COMMAND_ATTACHMENT_TYPE,
|
|
56
|
+
QUEUED_PROMPT_KEY,
|
|
57
|
+
RESULT_IS_ERROR_KEY,
|
|
58
|
+
RESULT_TOOL_USE_ID_KEY,
|
|
59
|
+
SUBAGENT_TYPE_INPUT_KEY,
|
|
60
|
+
TEXT_BLOCK_TYPE,
|
|
61
|
+
TEXT_KEY,
|
|
62
|
+
TOOL_RESULT_BLOCK_TYPE,
|
|
63
|
+
TOOL_USE_BLOCK_TYPE,
|
|
64
|
+
USER_ENTRY_TYPE,
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class SessionStep(enum.Enum):
|
|
69
|
+
"""One transcript step the readiness checks count."""
|
|
70
|
+
|
|
71
|
+
USER_MESSAGE = "user_message"
|
|
72
|
+
READ = "read"
|
|
73
|
+
QUESTION = "question"
|
|
74
|
+
OTHER_TOOL_CALL = "other_tool_call"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _is_user_text(text: str) -> bool:
|
|
78
|
+
if not text.strip():
|
|
79
|
+
return False
|
|
80
|
+
if any(each_marker in text for each_marker in ALL_HARNESS_ENVELOPE_MARKERS):
|
|
81
|
+
return HUMAN_SENDER_MARKER in text
|
|
82
|
+
return True
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _mcp_step(action_name: str) -> SessionStep:
|
|
86
|
+
if action_name in ALL_INTERACTIVE_MCP_ACTIONS:
|
|
87
|
+
return SessionStep.QUESTION
|
|
88
|
+
if action_name.startswith(ALL_MCP_READ_VERB_PREFIXES):
|
|
89
|
+
return SessionStep.READ
|
|
90
|
+
return SessionStep.OTHER_TOOL_CALL
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _tool_use_step(tool_name: str, tool_input: object) -> SessionStep:
|
|
94
|
+
all_input_fields = tool_input if isinstance(tool_input, dict) else {}
|
|
95
|
+
if tool_name in ALL_INTERACTIVE_QUESTION_TOOL_NAMES:
|
|
96
|
+
return SessionStep.QUESTION
|
|
97
|
+
if tool_name in ALL_READ_TOOL_NAMES:
|
|
98
|
+
return SessionStep.READ
|
|
99
|
+
if (
|
|
100
|
+
tool_name in ALL_BRIEF_FIELDS_BY_SPAWN_TOOL_NAME
|
|
101
|
+
and all_input_fields.get(SUBAGENT_TYPE_INPUT_KEY) in ALL_READ_ONLY_SUBAGENT_TYPES
|
|
102
|
+
):
|
|
103
|
+
return SessionStep.READ
|
|
104
|
+
if tool_name.startswith(MCP_TOOL_NAME_PREFIX):
|
|
105
|
+
return _mcp_step(tool_name.rsplit(MCP_TOOL_NAME_SEPARATOR, 1)[-1])
|
|
106
|
+
return SessionStep.OTHER_TOOL_CALL
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _content_of(all_entry_fields: dict[object, object]) -> object:
|
|
110
|
+
message = all_entry_fields.get(MESSAGE_KEY)
|
|
111
|
+
return message.get(CONTENT_KEY) if isinstance(message, dict) else None
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _assistant_steps(
|
|
115
|
+
all_entry_fields: dict[object, object],
|
|
116
|
+
all_question_tool_use_ids: set[str],
|
|
117
|
+
spawn_tool_use_id: str | None,
|
|
118
|
+
) -> Iterator[SessionStep]:
|
|
119
|
+
content = _content_of(all_entry_fields)
|
|
120
|
+
all_blocks = content if isinstance(content, list) else []
|
|
121
|
+
for each_block in all_blocks:
|
|
122
|
+
if (
|
|
123
|
+
not isinstance(each_block, dict)
|
|
124
|
+
or each_block.get(BLOCK_TYPE_KEY) != TOOL_USE_BLOCK_TYPE
|
|
125
|
+
):
|
|
126
|
+
continue
|
|
127
|
+
block_id = each_block.get(BLOCK_ID_KEY)
|
|
128
|
+
tool_name = each_block.get(BLOCK_NAME_KEY)
|
|
129
|
+
if block_id == spawn_tool_use_id or not isinstance(tool_name, str):
|
|
130
|
+
continue
|
|
131
|
+
if tool_name in ALL_INTERACTIVE_QUESTION_TOOL_NAMES and isinstance(block_id, str):
|
|
132
|
+
all_question_tool_use_ids.add(block_id)
|
|
133
|
+
yield _tool_use_step(tool_name, each_block.get(BLOCK_INPUT_KEY))
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _is_user_block(
|
|
137
|
+
all_block_fields: dict[object, object], all_question_tool_use_ids: set[str]
|
|
138
|
+
) -> bool:
|
|
139
|
+
block_type = all_block_fields.get(BLOCK_TYPE_KEY)
|
|
140
|
+
if block_type == TEXT_BLOCK_TYPE:
|
|
141
|
+
return _is_user_text(str(all_block_fields.get(TEXT_KEY, "")))
|
|
142
|
+
return (
|
|
143
|
+
block_type == TOOL_RESULT_BLOCK_TYPE
|
|
144
|
+
and all_block_fields.get(RESULT_TOOL_USE_ID_KEY) in all_question_tool_use_ids
|
|
145
|
+
and not all_block_fields.get(RESULT_IS_ERROR_KEY)
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _user_steps(
|
|
150
|
+
all_entry_fields: dict[object, object], all_question_tool_use_ids: set[str]
|
|
151
|
+
) -> Iterator[SessionStep]:
|
|
152
|
+
if all_entry_fields.get(IS_META_KEY) or all_entry_fields.get(IS_COMPACT_SUMMARY_KEY):
|
|
153
|
+
return
|
|
154
|
+
content = _content_of(all_entry_fields)
|
|
155
|
+
if isinstance(content, str):
|
|
156
|
+
is_user_message = _is_user_text(content)
|
|
157
|
+
else:
|
|
158
|
+
all_blocks = content if isinstance(content, list) else []
|
|
159
|
+
is_user_message = any(
|
|
160
|
+
isinstance(each_block, dict) and _is_user_block(each_block, all_question_tool_use_ids)
|
|
161
|
+
for each_block in all_blocks
|
|
162
|
+
)
|
|
163
|
+
if is_user_message:
|
|
164
|
+
yield SessionStep.USER_MESSAGE
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _queued_steps(all_entry_fields: dict[object, object]) -> Iterator[SessionStep]:
|
|
168
|
+
attachment = all_entry_fields.get(ATTACHMENT_KEY)
|
|
169
|
+
if (
|
|
170
|
+
not isinstance(attachment, dict)
|
|
171
|
+
or attachment.get(ENTRY_TYPE_KEY) != QUEUED_COMMAND_ATTACHMENT_TYPE
|
|
172
|
+
):
|
|
173
|
+
return
|
|
174
|
+
queued_prompt = attachment.get(QUEUED_PROMPT_KEY)
|
|
175
|
+
if isinstance(queued_prompt, str) and _is_user_text(queued_prompt):
|
|
176
|
+
yield SessionStep.USER_MESSAGE
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _parsed_entry(transcript_line: str) -> dict[object, object]:
|
|
180
|
+
try:
|
|
181
|
+
parsed_entry = json.loads(transcript_line)
|
|
182
|
+
except json.JSONDecodeError:
|
|
183
|
+
return {}
|
|
184
|
+
return parsed_entry if isinstance(parsed_entry, dict) else {}
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _entry_steps(
|
|
188
|
+
transcript_line: str, all_question_tool_use_ids: set[str], spawn_tool_use_id: str | None
|
|
189
|
+
) -> Iterator[SessionStep]:
|
|
190
|
+
parsed_entry = _parsed_entry(transcript_line)
|
|
191
|
+
entry_type = parsed_entry.get(ENTRY_TYPE_KEY)
|
|
192
|
+
if entry_type == ASSISTANT_ENTRY_TYPE:
|
|
193
|
+
return _assistant_steps(parsed_entry, all_question_tool_use_ids, spawn_tool_use_id)
|
|
194
|
+
if entry_type == USER_ENTRY_TYPE:
|
|
195
|
+
return _user_steps(parsed_entry, all_question_tool_use_ids)
|
|
196
|
+
if entry_type == ATTACHMENT_ENTRY_TYPE:
|
|
197
|
+
return _queued_steps(parsed_entry)
|
|
198
|
+
return iter(())
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def session_steps(
|
|
202
|
+
all_transcript_lines: Iterable[str], spawn_tool_use_id: str | None = None
|
|
203
|
+
) -> list[SessionStep]:
|
|
204
|
+
"""Read the transcript into the user messages and tool calls it holds, in order.
|
|
205
|
+
|
|
206
|
+
A tool call that is neither a read nor a question is an OTHER_TOOL_CALL,
|
|
207
|
+
so two user messages count as one only when nothing ran between them.
|
|
208
|
+
|
|
209
|
+
Args:
|
|
210
|
+
all_transcript_lines: The session transcript, one JSON entry per line.
|
|
211
|
+
spawn_tool_use_id: The id of the spawn under check, left out of the steps.
|
|
212
|
+
"""
|
|
213
|
+
all_question_tool_use_ids: set[str] = set()
|
|
214
|
+
return [
|
|
215
|
+
each_step
|
|
216
|
+
for each_line in all_transcript_lines
|
|
217
|
+
for each_step in _entry_steps(each_line, all_question_tool_use_ids, spawn_tool_use_id)
|
|
218
|
+
]
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _request_position(all_steps: list[SessionStep]) -> int | None:
|
|
222
|
+
all_user_positions = [
|
|
223
|
+
each_position
|
|
224
|
+
for each_position, each_step in enumerate(all_steps)
|
|
225
|
+
if each_step is SessionStep.USER_MESSAGE
|
|
226
|
+
and (each_position == 0 or all_steps[each_position - 1] is not SessionStep.USER_MESSAGE)
|
|
227
|
+
]
|
|
228
|
+
if not all_user_positions:
|
|
229
|
+
return None
|
|
230
|
+
for each_index in range(len(all_user_positions) - 1, 0, -1):
|
|
231
|
+
previous_position = all_user_positions[each_index - 1]
|
|
232
|
+
current_position = all_user_positions[each_index]
|
|
233
|
+
if SessionStep.QUESTION not in all_steps[previous_position:current_position]:
|
|
234
|
+
return current_position
|
|
235
|
+
return all_user_positions[0]
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def readiness_gaps(all_steps: list[SessionStep]) -> list[str]:
|
|
239
|
+
"""Return the deny reasons the session has not yet cleared, in order.
|
|
240
|
+
|
|
241
|
+
::
|
|
242
|
+
|
|
243
|
+
[USER_MESSAGE, READ, QUESTION, USER_MESSAGE] -> []
|
|
244
|
+
[USER_MESSAGE, QUESTION, USER_MESSAGE] -> [investigation reason]
|
|
245
|
+
[USER_MESSAGE, READ] -> [interview reason]
|
|
246
|
+
[USER_MESSAGE, READ, QUESTION] -> [interview reason]
|
|
247
|
+
|
|
248
|
+
The span starts at the request: the latest user message that answers no
|
|
249
|
+
question. The interview passes when a user message in that span answers
|
|
250
|
+
a question.
|
|
251
|
+
|
|
252
|
+
Args:
|
|
253
|
+
all_steps: The steps from ``session_steps``.
|
|
254
|
+
"""
|
|
255
|
+
request_position = _request_position(all_steps)
|
|
256
|
+
span_steps = all_steps if request_position is None else all_steps[request_position:]
|
|
257
|
+
all_gaps: list[str] = []
|
|
258
|
+
if SessionStep.READ not in span_steps:
|
|
259
|
+
all_gaps.append(MISSING_INVESTIGATION_REASON)
|
|
260
|
+
has_reply = any(
|
|
261
|
+
each_step is SessionStep.USER_MESSAGE and SessionStep.QUESTION in span_steps[:each_position]
|
|
262
|
+
for each_position, each_step in enumerate(span_steps)
|
|
263
|
+
)
|
|
264
|
+
if not has_reply:
|
|
265
|
+
all_gaps.append(MISSING_INTERVIEW_REASON)
|
|
266
|
+
return all_gaps
|