@ethlete/agent-rules 0.1.0-next.13 → 0.1.0-next.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +73 -0
- package/README.md +36 -2
- package/THIRD-PARTY-LICENSES.md +35 -0
- package/content/hooks/context-warning.py +450 -112
- package/content/hooks/subagent-model-policy.py +185 -0
- package/content/rules/subagent-models.md +29 -0
- package/content/skills/codex-subagent/SKILL.md +90 -0
- package/content/skills/codex-subagent/codex-agent.mjs +229 -0
- package/content/skills/design-exploration/SKILL.md +203 -0
- package/content/skills/design-exploration/check-story.mjs +139 -0
- package/content/skills/design-exploration/shoot-template.mjs +61 -0
- package/content/skills/domain-modeling/ADR-FORMAT.md +47 -0
- package/content/skills/domain-modeling/CONTEXT-FORMAT.md +60 -0
- package/content/skills/domain-modeling/SKILL.md +80 -0
- package/content/skills/grill-with-docs/SKILL.md +13 -0
- package/content/skills/grilling/SKILL.md +34 -0
- package/content/skills/handoff/SKILL.md +69 -4
- package/content/skills/sdk-update/SKILL.md +89 -0
- package/content/skills/timetrack/SKILL.md +190 -10
- package/package.json +1 -1
- package/src/index.js +2 -1
- package/src/index.js.map +1 -1
- package/src/lib/frontmatter.d.ts +2 -0
- package/src/lib/frontmatter.js +2 -1
- package/src/lib/frontmatter.js.map +1 -1
- package/src/lib/git-flow/config.js +1 -1
- package/src/lib/git-flow/config.js.map +1 -1
- package/src/lib/git.d.ts +37 -0
- package/src/lib/git.js +77 -1
- package/src/lib/git.js.map +1 -1
- package/src/lib/index.d.ts +1 -0
- package/src/lib/index.js +1 -0
- package/src/lib/index.js.map +1 -1
- package/src/lib/plain-text.d.ts +9 -0
- package/src/lib/plain-text.js +22 -0
- package/src/lib/plain-text.js.map +1 -0
- package/src/lib/targets/claude-hooks.js +2 -1
- package/src/lib/targets/claude-hooks.js.map +1 -1
- package/src/lib/targets/codex-hooks.js +2 -1
- package/src/lib/targets/codex-hooks.js.map +1 -1
- package/src/lib/targets/hooks-shared.d.ts +22 -1
- package/src/lib/targets/hooks-shared.js +42 -9
- package/src/lib/targets/hooks-shared.js.map +1 -1
- package/src/lib/targets/shared.js +1 -0
- package/src/lib/targets/shared.js.map +1 -1
- package/src/lib/timetrack-command.js +449 -20
- package/src/lib/timetrack-command.js.map +1 -1
- package/src/lib/timetrack.d.ts +241 -0
- package/src/lib/timetrack.js +83 -2
- package/src/lib/timetrack.js.map +1 -1
|
@@ -1,21 +1,40 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
|
-
"""
|
|
2
|
+
"""Context-budget hook: keep the agent aware of how much context it may still spend.
|
|
3
3
|
|
|
4
4
|
Reads the hook input JSON from stdin, estimates the current context size from
|
|
5
5
|
the session transcript, and emits a warning (visible to both the user and the
|
|
6
6
|
agent) when it crosses a threshold. Recommends the handoff skill so work can
|
|
7
7
|
continue in a fresh session.
|
|
8
8
|
|
|
9
|
+
Registered on four events, because a warning the agent cannot see while it works
|
|
10
|
+
is not a warning:
|
|
11
|
+
|
|
12
|
+
* SessionStart - states the budget and how to scope work to it, before the
|
|
13
|
+
first request. Nothing else tells the model what its budget is.
|
|
14
|
+
* UserPromptSubmit - the tiered warning, plus a short reminder on every later
|
|
15
|
+
prompt.
|
|
16
|
+
* PostToolBatch - the same tiered warning, delivered mid-run. This is the only
|
|
17
|
+
event that fires between an agent's own model requests, so without it a long
|
|
18
|
+
autonomous run never learns its token count until the user types something.
|
|
19
|
+
* Stop - fires as the agent tries to end its turn. Its additionalContext is
|
|
20
|
+
delivered to the model and the conversation continues, so the decision to
|
|
21
|
+
finish or hand off happens without the user having to ask for it.
|
|
22
|
+
|
|
9
23
|
Runs under both Claude Code and Codex, selected by `--agent` on the command
|
|
10
|
-
line rather than sniffed from the payload
|
|
11
|
-
the two never have to be told apart at runtime.
|
|
12
|
-
|
|
24
|
+
line rather than sniffed from the payload - the generator writes the flag, so
|
|
25
|
+
the two never have to be told apart at runtime. Codex has only UserPromptSubmit;
|
|
26
|
+
the other three events are registered for Claude alone. The two differ in three
|
|
27
|
+
ways that matter here, all captured in AGENT_PROFILES:
|
|
13
28
|
|
|
14
29
|
* Transcript format. Claude writes one JSON object per message with
|
|
15
30
|
`message.usage`; Codex writes a rollout stream whose `token_count` events
|
|
16
31
|
carry a TokenUsageInfo. Codex's rollout format is explicitly not a stable
|
|
17
32
|
interface, so the parser searches each line for the usage object instead of
|
|
18
33
|
walking a fixed path.
|
|
34
|
+
* Sub-agents. Codex gives each sub-agent its own rollout and thread id while
|
|
35
|
+
retaining the root session id. Warnings identify that thread explicitly and
|
|
36
|
+
tell it to report back to its parent instead of making the main agent hand
|
|
37
|
+
off its session.
|
|
19
38
|
* Context budget. Claude's budget is capped at its 200k long-context pricing
|
|
20
39
|
boundary. Codex uses the model-specific pricing boundary where OpenAI
|
|
21
40
|
documents one, and the reported window otherwise. Crossing either boundary
|
|
@@ -23,7 +42,7 @@ that matter here, all captured in AGENT_PROFILES:
|
|
|
23
42
|
* How a handoff is invoked. Claude has a slash command and /clear; Codex has
|
|
24
43
|
neither and reads the skill from disk. The skill's own name differs per repo
|
|
25
44
|
(`ethlete-handoff` where the generator installed it, `handoff` where the repo
|
|
26
|
-
ships its own copy), so it is resolved at runtime
|
|
45
|
+
ships its own copy), so it is resolved at runtime - see handoff_skill().
|
|
27
46
|
|
|
28
47
|
At the critical tier, if the session's permission_mode is "auto", the
|
|
29
48
|
instruction escalates from "recommend" to "just do it": auto mode already
|
|
@@ -34,8 +53,21 @@ handoff file itself immediately rather than waiting for a natural stopping
|
|
|
34
53
|
point. Codex's permission_mode value set is undocumented, so its profile lists
|
|
35
54
|
no auto modes and the escalation stays off there.
|
|
36
55
|
|
|
56
|
+
Neither the critical tier nor the Stop event forces a handoff. A session is
|
|
57
|
+
often a step or two from done when they fire, and some sessions hold reasoning
|
|
58
|
+
that no handoff file survives - crossing the boundary is then the cheaper
|
|
59
|
+
choice. So the Stop event offers three options and demands the agent name the
|
|
60
|
+
one it took: finish, hand off, or continue past the boundary for a reason it
|
|
61
|
+
states in one sentence. A third tier at 95% withdraws only the "finish" option,
|
|
62
|
+
because a budget that small covers no task.
|
|
63
|
+
|
|
37
64
|
Warns once per tier per session (state kept in a temp file); re-arms itself
|
|
38
|
-
if the context shrinks again (e.g. after a compaction).
|
|
65
|
+
if the context shrinks again (e.g. after a compaction). The Stop event keeps its
|
|
66
|
+
own per-tier counter, so crossing a tier mid-run costs at most one notice while
|
|
67
|
+
working and one demand at the end of the turn. Every prompt after the first
|
|
68
|
+
warning carries a short reminder instead: how much budget is left, and how to
|
|
69
|
+
work so that either ending stays cheap. One warning at 70% is stale by the time
|
|
70
|
+
it matters, and the agent cannot see its own token count.
|
|
39
71
|
|
|
40
72
|
Can be disabled per machine via a gitignored ethlete-agents.config.local.json
|
|
41
73
|
at the repo root: {"disableHooks": true} or {"disableHooks": ["context-warning"]}.
|
|
@@ -43,7 +75,8 @@ To keep the tiered warnings but drop just the auto-mode auto-save escalation,
|
|
|
43
75
|
use {"disableAutoHandoffSave": true} instead - the critical tier then falls
|
|
44
76
|
back to recommending a handoff, same as non-auto mode.
|
|
45
77
|
|
|
46
|
-
Fail-safe: any error exits 0 with no output
|
|
78
|
+
Fail-safe: any error exits 0 with no output - the hook must never block a prompt,
|
|
79
|
+
a tool batch or the end of a turn.
|
|
47
80
|
"""
|
|
48
81
|
|
|
49
82
|
import json
|
|
@@ -54,12 +87,38 @@ import tempfile
|
|
|
54
87
|
HOOK_NAME = "context-warning"
|
|
55
88
|
LOCAL_CONFIG_FILE = "ethlete-agents.config.local.json"
|
|
56
89
|
|
|
57
|
-
# Warn / critical fire at these fractions of the token budget.
|
|
90
|
+
# Warn / critical / final fire at these fractions of the token budget.
|
|
58
91
|
WARN_FRACTION = 0.70
|
|
59
92
|
CRITICAL_FRACTION = 0.85
|
|
93
|
+
FINAL_FRACTION = 0.95
|
|
94
|
+
|
|
95
|
+
# How to work once the budget is tight - repeated on every prompt from the first
|
|
96
|
+
# warning on, so it stays in view as the budget shrinks.
|
|
97
|
+
HANDOFF_MODE = (
|
|
98
|
+
"Work in slices that each end in a committable state. Send bulky reads, searches and "
|
|
99
|
+
"test runs to a sub-agent so their output stays out of this context. Write down each "
|
|
100
|
+
"decision and dead end as you reach it, in the commit message or the handoff file. "
|
|
101
|
+
"Start nothing you cannot finish or hand off inside the remaining budget."
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
# Said once at session start, before anything has been read - the only point at which the
|
|
105
|
+
# scope of the whole session is still open.
|
|
106
|
+
SESSION_BUDGET = (
|
|
107
|
+
"Scope the session to it. Send bulky reads, searches and test runs to a sub-agent so "
|
|
108
|
+
"their output never enters this context. Commit in slices. A task that plainly does not "
|
|
109
|
+
"fit should be split now, not abandoned at 90%."
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
# The escape from a handoff at the critical tier, with the test that keeps it honest.
|
|
113
|
+
FINISH_FIRST = (
|
|
114
|
+
"If you can name every step that is left, and they plainly fit the remaining budget, "
|
|
115
|
+
"finish the task, commit, and then apply the handoff skill's own save test - with "
|
|
116
|
+
"everything committed and no step left, write no handoff file and say so instead. "
|
|
117
|
+
"\"Nearly done\" means you can name the last steps now, not that the end feels close."
|
|
118
|
+
)
|
|
60
119
|
|
|
61
120
|
# On models with a window larger than this, tokens beyond it bill the whole
|
|
62
|
-
# context at the long-context premium rate
|
|
121
|
+
# context at the long-context premium rate - so the budget never exceeds it.
|
|
63
122
|
PREMIUM_BOUNDARY = 200_000
|
|
64
123
|
|
|
65
124
|
# Codex models whose long-context pricing starts above 272k input tokens. Models
|
|
@@ -72,7 +131,7 @@ CODEX_PREMIUM_BOUNDARIES = (
|
|
|
72
131
|
)
|
|
73
132
|
|
|
74
133
|
# Context window (tokens) per model, matched by substring against the model id
|
|
75
|
-
# from the transcript
|
|
134
|
+
# from the transcript - first match wins. Edit these as model windows change;
|
|
76
135
|
# anything unmatched falls back to DEFAULT_WINDOW.
|
|
77
136
|
CONTEXT_WINDOWS = (
|
|
78
137
|
("opus-5", 1_000_000),
|
|
@@ -80,7 +139,7 @@ CONTEXT_WINDOWS = (
|
|
|
80
139
|
("sonnet-4-5", 1_000_000),
|
|
81
140
|
("sonnet-5", 1_000_000),
|
|
82
141
|
("fable-5", 1_000_000),
|
|
83
|
-
# Generic fallbacks
|
|
142
|
+
# Generic fallbacks - keep these last: the first substring match wins, so a bare
|
|
84
143
|
# "opus"/"sonnet" entry placed above would shadow every versioned entry below it.
|
|
85
144
|
("opus", 200_000),
|
|
86
145
|
("sonnet", 200_000),
|
|
@@ -88,6 +147,13 @@ CONTEXT_WINDOWS = (
|
|
|
88
147
|
)
|
|
89
148
|
DEFAULT_WINDOW = 200_000
|
|
90
149
|
|
|
150
|
+
# Bytes read from the end of a Claude transcript to find the newest usage report. Read on
|
|
151
|
+
# every tool batch, so a full parse of a multi-megabyte transcript would add latency to
|
|
152
|
+
# every model request. Grows by GROWTH_FACTOR until a usage report is found.
|
|
153
|
+
TAIL_BYTES = 256 * 1024
|
|
154
|
+
MAX_TAIL_BYTES = 16 * 1024 * 1024
|
|
155
|
+
GROWTH_FACTOR = 8
|
|
156
|
+
|
|
91
157
|
AGENT_PROFILES = {
|
|
92
158
|
"claude": {
|
|
93
159
|
"premium_boundary": PREMIUM_BOUNDARY,
|
|
@@ -101,11 +167,17 @@ AGENT_PROFILES = {
|
|
|
101
167
|
"recommend": "recommend the user run /{handoff} to save state and start a fresh session",
|
|
102
168
|
"suggest": "suggest the user run /{handoff} to save state and start a fresh session",
|
|
103
169
|
"save_now": (
|
|
104
|
-
"run the handoff skill's save mode right now
|
|
105
|
-
"edit first, nothing new
|
|
106
|
-
"
|
|
107
|
-
"
|
|
108
|
-
"
|
|
170
|
+
"run the handoff skill's save mode right now - finish only an in-flight atomic "
|
|
171
|
+
"edit first, nothing new. Apply its own test: with every change committed, no "
|
|
172
|
+
"step left and nothing running, write no file, say that instead, and recommend "
|
|
173
|
+
"/clear. If a file is warranted, tell the user exactly which one was written and "
|
|
174
|
+
"that they should run /clear, then '/{handoff} resume <slug>', to continue - "
|
|
175
|
+
"clearing and resuming can't be done programmatically, so this is the one step "
|
|
176
|
+
"still on them."
|
|
177
|
+
),
|
|
178
|
+
"save_at_stop": (
|
|
179
|
+
"save a handoff with the handoff skill, then tell the user which file was "
|
|
180
|
+
"written and that they should run /clear, then '/{handoff} resume <slug>'"
|
|
109
181
|
),
|
|
110
182
|
},
|
|
111
183
|
"codex": {
|
|
@@ -132,10 +204,15 @@ AGENT_PROFILES = {
|
|
|
132
204
|
"continuing in a fresh session"
|
|
133
205
|
),
|
|
134
206
|
"save_now": (
|
|
135
|
-
"follow .agents/skills/{handoff}/SKILL.md and save a handoff right now "
|
|
136
|
-
"
|
|
137
|
-
"
|
|
138
|
-
"
|
|
207
|
+
"follow .agents/skills/{handoff}/SKILL.md and save a handoff right now - finish "
|
|
208
|
+
"only an in-flight atomic edit first, nothing new. Apply its own test: with "
|
|
209
|
+
"every change committed, no step left and nothing running, write no file and say "
|
|
210
|
+
"that instead. If a file is warranted, tell the user exactly which one was "
|
|
211
|
+
"written and that they should start a fresh codex session and resume from it."
|
|
212
|
+
),
|
|
213
|
+
"save_at_stop": (
|
|
214
|
+
"save a handoff via .agents/skills/{handoff}/SKILL.md, then tell the user which "
|
|
215
|
+
"file was written and that a fresh session should resume from it"
|
|
139
216
|
),
|
|
140
217
|
},
|
|
141
218
|
}
|
|
@@ -145,7 +222,7 @@ def handoff_skill(root):
|
|
|
145
222
|
"""Name of the handoff skill installed in this repo.
|
|
146
223
|
|
|
147
224
|
The generator prefixes every skill it writes, so the command is /ethlete-handoff in a
|
|
148
|
-
consumer repo
|
|
225
|
+
consumer repo - but a repo that excludes the generated copy and ships its own (the SDK
|
|
149
226
|
itself) has a plain /handoff instead. Naming the wrong one sends the user to a slash
|
|
150
227
|
command that does not exist, so it is read off disk rather than assumed.
|
|
151
228
|
"""
|
|
@@ -165,7 +242,7 @@ def resolve_profile(profile, skill):
|
|
|
165
242
|
|
|
166
243
|
|
|
167
244
|
def agent_name(argv):
|
|
168
|
-
"""The --agent value, defaulting to claude
|
|
245
|
+
"""The --agent value, defaulting to claude - older registrations pass no flag."""
|
|
169
246
|
for index, arg in enumerate(argv):
|
|
170
247
|
if arg == "--agent" and index + 1 < len(argv):
|
|
171
248
|
return argv[index + 1]
|
|
@@ -233,35 +310,62 @@ def premium_boundary_for(profile, model):
|
|
|
233
310
|
return profile["premium_boundary"]
|
|
234
311
|
|
|
235
312
|
|
|
313
|
+
def tail_lines(path, limit):
|
|
314
|
+
"""(lines newest-first, whole file read) for the last `limit` bytes of a file.
|
|
315
|
+
|
|
316
|
+
A partial first line is dropped, so every returned line is complete.
|
|
317
|
+
"""
|
|
318
|
+
size = os.path.getsize(path)
|
|
319
|
+
start = max(size - limit, 0)
|
|
320
|
+
with open(path, "rb") as f:
|
|
321
|
+
f.seek(start)
|
|
322
|
+
chunk = f.read()
|
|
323
|
+
if start > 0:
|
|
324
|
+
newline = chunk.find(b"\n")
|
|
325
|
+
chunk = chunk[newline + 1 :] if newline != -1 else b""
|
|
326
|
+
return list(reversed(chunk.split(b"\n"))), start == 0
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def claude_usage_in(lines):
|
|
330
|
+
"""(usage, model) from the newest main-chain assistant message in `lines`, or (None, None)."""
|
|
331
|
+
for line in lines:
|
|
332
|
+
if not line.strip():
|
|
333
|
+
continue
|
|
334
|
+
try:
|
|
335
|
+
obj = json.loads(line)
|
|
336
|
+
except (json.JSONDecodeError, UnicodeDecodeError):
|
|
337
|
+
continue
|
|
338
|
+
if obj.get("type") != "assistant" or obj.get("isSidechain"):
|
|
339
|
+
continue
|
|
340
|
+
message = obj.get("message") or {}
|
|
341
|
+
usage = message.get("usage")
|
|
342
|
+
if usage:
|
|
343
|
+
return usage, message.get("model")
|
|
344
|
+
return None, None
|
|
345
|
+
|
|
346
|
+
|
|
236
347
|
def claude_context_state(transcript_path):
|
|
237
|
-
"""(tokens, model, window) from the
|
|
348
|
+
"""(tokens, model, window, thread id, is sub-agent) from the main-chain assistant message.
|
|
238
349
|
|
|
239
|
-
tokens
|
|
240
|
-
window in the transcript, so it is resolved from the model id by the caller.
|
|
350
|
+
tokens is its total input tokens (fresh + cache read + cache creation). Claude reports no
|
|
351
|
+
window in the transcript, so it is resolved from the model id by the caller. Only the tail
|
|
352
|
+
of the transcript is read, growing until a usage report is found - this runs on every tool
|
|
353
|
+
batch, and a transcript reaches tens of megabytes.
|
|
241
354
|
"""
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
last_model = message.get("model") or last_model
|
|
257
|
-
if not last_usage:
|
|
258
|
-
return 0, last_model, None
|
|
259
|
-
tokens = (
|
|
260
|
-
last_usage.get("input_tokens", 0)
|
|
261
|
-
+ last_usage.get("cache_read_input_tokens", 0)
|
|
262
|
-
+ last_usage.get("cache_creation_input_tokens", 0)
|
|
263
|
-
)
|
|
264
|
-
return tokens, last_model, None
|
|
355
|
+
limit = TAIL_BYTES
|
|
356
|
+
while True:
|
|
357
|
+
lines, complete = tail_lines(transcript_path, limit)
|
|
358
|
+
usage, model = claude_usage_in(lines)
|
|
359
|
+
if usage:
|
|
360
|
+
tokens = (
|
|
361
|
+
usage.get("input_tokens", 0)
|
|
362
|
+
+ usage.get("cache_read_input_tokens", 0)
|
|
363
|
+
+ usage.get("cache_creation_input_tokens", 0)
|
|
364
|
+
)
|
|
365
|
+
return tokens, model, None, None, False
|
|
366
|
+
if complete or limit >= MAX_TAIL_BYTES:
|
|
367
|
+
return 0, None, None, None, False
|
|
368
|
+
limit *= GROWTH_FACTOR
|
|
265
369
|
|
|
266
370
|
|
|
267
371
|
def find_token_usage_info(node):
|
|
@@ -286,20 +390,29 @@ def find_token_usage_info(node):
|
|
|
286
390
|
|
|
287
391
|
|
|
288
392
|
def codex_context_state(transcript_path):
|
|
289
|
-
"""(tokens, model, window
|
|
393
|
+
"""(tokens, model, window, thread id, is sub-agent) from a Codex rollout.
|
|
290
394
|
|
|
291
|
-
tokens is the last request's `total_tokens`
|
|
395
|
+
tokens is the last request's `total_tokens` - Codex's own `tokens_in_context_window()`
|
|
292
396
|
is exactly that field, and `last_token_usage` is replaced per request while
|
|
293
397
|
`total_token_usage` accumulates across the whole session.
|
|
294
398
|
"""
|
|
295
399
|
last_info = None
|
|
296
400
|
last_model = None
|
|
401
|
+
thread_id = None
|
|
402
|
+
is_subagent = False
|
|
297
403
|
with open(transcript_path, encoding="utf-8") as f:
|
|
298
404
|
for line in f:
|
|
299
405
|
try:
|
|
300
406
|
obj = json.loads(line)
|
|
301
407
|
except json.JSONDecodeError:
|
|
302
408
|
continue
|
|
409
|
+
if obj.get("type") == "session_meta":
|
|
410
|
+
payload = obj.get("payload") or {}
|
|
411
|
+
source = payload.get("source")
|
|
412
|
+
thread_id = payload.get("id") or thread_id
|
|
413
|
+
is_subagent = payload.get("thread_source") == "subagent" or (
|
|
414
|
+
isinstance(source, dict) and isinstance(source.get("subagent"), dict)
|
|
415
|
+
)
|
|
303
416
|
model = (obj.get("payload") or {}).get("model") if isinstance(obj.get("payload"), dict) else None
|
|
304
417
|
if isinstance(model, str) and model:
|
|
305
418
|
last_model = model
|
|
@@ -307,72 +420,265 @@ def codex_context_state(transcript_path):
|
|
|
307
420
|
if info:
|
|
308
421
|
last_info = info
|
|
309
422
|
if not last_info:
|
|
310
|
-
return 0, last_model, None
|
|
423
|
+
return 0, last_model, None, thread_id, is_subagent
|
|
311
424
|
usage = last_info.get("last_token_usage") or {}
|
|
312
425
|
window = last_info.get("model_context_window")
|
|
313
|
-
return
|
|
426
|
+
return (
|
|
427
|
+
usage.get("total_tokens", 0),
|
|
428
|
+
last_model,
|
|
429
|
+
window if isinstance(window, int) else None,
|
|
430
|
+
thread_id,
|
|
431
|
+
is_subagent,
|
|
432
|
+
)
|
|
314
433
|
|
|
315
434
|
|
|
316
435
|
CONTEXT_READERS = {"claude": claude_context_state, "codex": codex_context_state}
|
|
317
436
|
|
|
318
437
|
|
|
438
|
+
def headroom(tokens, budget):
|
|
439
|
+
"""How much budget is left, as a short human string."""
|
|
440
|
+
left = max(budget - tokens, 0)
|
|
441
|
+
return f"~{left // 1000}k tokens" if left >= 1000 else "under 1k tokens"
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def limit_name(priced):
|
|
445
|
+
return "long-context pricing boundary" if priced else "context window"
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def cost_of_crossing(priced):
|
|
449
|
+
"""What the user pays for one more token past the budget."""
|
|
450
|
+
if priced:
|
|
451
|
+
return "every later request is billed at the premium rate"
|
|
452
|
+
return "auto-compact fires and the detail of this session is summarised away"
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
def token_count(tokens):
|
|
456
|
+
"""A token count as a short label: 1M rather than 1000k."""
|
|
457
|
+
if tokens >= 1_000_000 and tokens % 1_000_000 == 0:
|
|
458
|
+
return f"{tokens // 1_000_000}M"
|
|
459
|
+
return f"{tokens // 1000}k"
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def session_start_message(budget, window, model):
|
|
463
|
+
"""The one line said before the first request, naming the budget and how to fit it.
|
|
464
|
+
|
|
465
|
+
A session that starts fresh has no assistant message yet, so the model - and with it the
|
|
466
|
+
window - is unknown. The budget is not: it is the pricing boundary either way.
|
|
467
|
+
"""
|
|
468
|
+
budget_label = token_count(budget)
|
|
469
|
+
if not model:
|
|
470
|
+
frame = (
|
|
471
|
+
"past it a model with a larger window bills every request at the premium rate, and a "
|
|
472
|
+
f"{budget_label} model auto-compacts the session away"
|
|
473
|
+
)
|
|
474
|
+
elif budget < window:
|
|
475
|
+
frame = (
|
|
476
|
+
f"the long-context pricing boundary of this model's {token_count(window)} window; "
|
|
477
|
+
"past it every request is billed at the premium rate"
|
|
478
|
+
)
|
|
479
|
+
else:
|
|
480
|
+
frame = "this model's whole context window; past it auto-compact summarises the session away"
|
|
481
|
+
return (
|
|
482
|
+
f"[context-warning hook] Context budget for this session: {budget_label} tokens - {frame}. "
|
|
483
|
+
f"{SESSION_BUDGET} You will be warned at 70%, 85% and 95%."
|
|
484
|
+
)
|
|
485
|
+
|
|
486
|
+
|
|
319
487
|
def messages(profile, tier, tokens, budget, priced, auto_mode):
|
|
320
488
|
"""(systemMessage, additionalContext) for a tier.
|
|
321
489
|
|
|
322
|
-
priced: the budget is the pricing boundary, not the window
|
|
490
|
+
priced: the budget is the pricing boundary, not the window - the reason to
|
|
323
491
|
hand off is cost, not an imminent auto-compact.
|
|
324
|
-
auto_mode:
|
|
325
|
-
"save it now"
|
|
492
|
+
auto_mode: from the critical tier on, escalates from "recommend a handoff"
|
|
493
|
+
to "save it now" - see the module docstring for why.
|
|
326
494
|
"""
|
|
327
495
|
k = f"~{tokens // 1000}k"
|
|
328
496
|
pct = round(tokens / budget * 100)
|
|
329
497
|
budget_k = f"{budget // 1000}k"
|
|
498
|
+
left = headroom(tokens, budget)
|
|
330
499
|
|
|
331
500
|
if priced:
|
|
332
|
-
approach = "about to cross" if tier
|
|
333
|
-
headline = f"Context is at {k} tokens
|
|
501
|
+
approach = "about to cross" if tier >= 2 else "approaching"
|
|
502
|
+
headline = f"Context is at {k} tokens - {approach} the {budget_k} long-context pricing boundary"
|
|
334
503
|
detail = (
|
|
335
|
-
f"The context is at {k} tokens
|
|
336
|
-
f"pricing boundary."
|
|
504
|
+
f"The context is at {k} tokens - {pct}% of the {budget_k} long-context "
|
|
505
|
+
f"pricing boundary, {left} left."
|
|
337
506
|
)
|
|
507
|
+
pressure = "Past it every request is billed at the premium rate. "
|
|
508
|
+
tail = ""
|
|
338
509
|
else:
|
|
339
510
|
headline = f"Context is at {k} tokens ({pct}% of the {budget_k} window)"
|
|
340
511
|
detail = (
|
|
341
|
-
f"The context window is at {k} tokens
|
|
342
|
-
f"{budget_k} window"
|
|
512
|
+
f"The context window is at {k} tokens - {pct}% of this model's "
|
|
513
|
+
f"{budget_k} window, {left} left."
|
|
343
514
|
)
|
|
515
|
+
pressure = ""
|
|
516
|
+
tail = " - auto-compact is imminent" if tier >= 3 else " - auto-compact is close" if tier == 2 else ""
|
|
344
517
|
|
|
345
|
-
if tier
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
"Past it every request is billed at the premium rate. "
|
|
350
|
-
if priced
|
|
351
|
-
else ""
|
|
518
|
+
if tier >= 3:
|
|
519
|
+
gone = (
|
|
520
|
+
f"The finish-first choice from the last warning is gone: {left} does not "
|
|
521
|
+
f"cover another task."
|
|
352
522
|
)
|
|
353
|
-
tail = "" if priced else " — auto-compact is imminent"
|
|
354
523
|
if auto_mode:
|
|
355
524
|
return (
|
|
356
525
|
f"🔴 {headline}{tail}. Auto mode is active: saving a handoff now.",
|
|
526
|
+
f"[context-warning hook] {detail} {pressure}{gone} Auto mode is active, so "
|
|
527
|
+
f"{profile['save_now']}",
|
|
528
|
+
)
|
|
529
|
+
return (
|
|
530
|
+
f"🔴 {headline}{tail}. {profile['act_now']}",
|
|
531
|
+
f"[context-warning hook] {detail} {pressure}{gone} Stop after the current tool "
|
|
532
|
+
f"call, then {profile['recommend']}.",
|
|
533
|
+
)
|
|
534
|
+
|
|
535
|
+
if tier == 2:
|
|
536
|
+
if auto_mode:
|
|
537
|
+
return (
|
|
538
|
+
f"🔴 {headline}{tail}. Auto mode is active: finishing the task, or saving a handoff.",
|
|
357
539
|
f"[context-warning hook] {detail} {pressure}Auto mode is active, so don't "
|
|
358
|
-
f"just recommend a handoff
|
|
540
|
+
f"just recommend a handoff. Choose one of two, and tell the user which. "
|
|
541
|
+
f"{FINISH_FIRST} Otherwise {profile['save_now']}",
|
|
359
542
|
)
|
|
360
543
|
return (
|
|
361
544
|
f"🔴 {headline}{tail}. {profile['act_now']}",
|
|
362
|
-
f"[context-warning hook] {detail} {pressure}
|
|
363
|
-
f"
|
|
545
|
+
f"[context-warning hook] {detail} {pressure}Choose one of two, and tell the user "
|
|
546
|
+
f"which. {FINISH_FIRST} Otherwise finish only the immediate step, then "
|
|
547
|
+
f"{profile['recommend']}. Either way, start no new sub-tasks.",
|
|
364
548
|
)
|
|
365
549
|
|
|
366
|
-
threshold = f" (≥{int(WARN_FRACTION * 100)}%)"
|
|
367
|
-
detail = detail if priced else f"{detail}{threshold}."
|
|
368
|
-
pressure = "Past it every request is billed at the premium rate. " if priced else ""
|
|
369
550
|
return (
|
|
370
551
|
f"🟡 {headline}. At the next natural stopping point, {profile['act_later']}.",
|
|
371
|
-
f"[context-warning hook] {detail} {pressure}
|
|
372
|
-
f"
|
|
552
|
+
f"[context-warning hook] {detail} {pressure}Handoff mode is active from here on. "
|
|
553
|
+
f"{HANDOFF_MODE} Keep working; when the current task reaches a natural stopping "
|
|
554
|
+
f"point, {profile['suggest']}.",
|
|
555
|
+
)
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def stop_messages(profile, tier, tokens, budget, priced):
|
|
559
|
+
"""(systemMessage, additionalContext) for a turn that is about to end over budget.
|
|
560
|
+
|
|
561
|
+
The turn continues after this, so the agent acts on it without the user having to
|
|
562
|
+
send a message. It is the only point where the agent is between tasks and can still
|
|
563
|
+
spend tokens, so the choice is put here in full - including the option to cross the
|
|
564
|
+
boundary on purpose, which some sessions are worth.
|
|
565
|
+
"""
|
|
566
|
+
k = f"~{tokens // 1000}k"
|
|
567
|
+
pct = round(tokens / budget * 100)
|
|
568
|
+
budget_k = f"{budget // 1000}k"
|
|
569
|
+
left = headroom(tokens, budget)
|
|
570
|
+
options = [
|
|
571
|
+
f"Hand off: {profile['save_at_stop']}.",
|
|
572
|
+
(
|
|
573
|
+
"Cross the boundary on purpose: take this only for a reason you can name in one "
|
|
574
|
+
"sentence right now, for example the task is one step from done, or this session "
|
|
575
|
+
f"holds reasoning a handoff file would lose. Say the reason, say that "
|
|
576
|
+
f"{cost_of_crossing(priced)}, and keep working."
|
|
577
|
+
),
|
|
578
|
+
]
|
|
579
|
+
if tier < 3:
|
|
580
|
+
options.insert(0, f"Finish the task now: {FINISH_FIRST}")
|
|
581
|
+
numbered = " ".join(f"{index}. {text}" for index, text in enumerate(options, start=1))
|
|
582
|
+
withdrawn = (
|
|
583
|
+
""
|
|
584
|
+
if tier < 3
|
|
585
|
+
else f" The finish-first option is withdrawn at this tier: {left} does not cover another task."
|
|
586
|
+
)
|
|
587
|
+
return (
|
|
588
|
+
f"🔴 Turn ending at {k} tokens ({pct}% of the {budget_k} {limit_name(priced)}). "
|
|
589
|
+
f"Deciding whether to hand off.",
|
|
590
|
+
f"[context-warning hook] This turn is about to end at {k} tokens - {pct}% of the "
|
|
591
|
+
f"{budget_k} {limit_name(priced)}, {left} left. Past it {cost_of_crossing(priced)}."
|
|
592
|
+
f"{withdrawn} Do not end the turn without taking one of these, and tell the user "
|
|
593
|
+
f"which you took. {numbered}",
|
|
373
594
|
)
|
|
374
595
|
|
|
375
596
|
|
|
597
|
+
def reminder(tier, tokens, budget, priced):
|
|
598
|
+
"""The short note repeated on every prompt after a tier was already warned about.
|
|
599
|
+
|
|
600
|
+
A single warning at 70% is stale by the time the budget is actually tight, and the
|
|
601
|
+
agent has no way to read its own token count between prompts.
|
|
602
|
+
"""
|
|
603
|
+
stop = (
|
|
604
|
+
" You were already told to hand off. Do not take on new work."
|
|
605
|
+
if tier >= 2
|
|
606
|
+
else ""
|
|
607
|
+
)
|
|
608
|
+
return (
|
|
609
|
+
f"[context-warning hook] Handoff mode is active: {headroom(tokens, budget)} left of "
|
|
610
|
+
f"the {budget // 1000}k {limit_name(priced)}. {HANDOFF_MODE}{stop}"
|
|
611
|
+
)
|
|
612
|
+
|
|
613
|
+
|
|
614
|
+
def subagent_message(tier, tokens, budget, priced):
|
|
615
|
+
k = f"~{tokens // 1000}k"
|
|
616
|
+
pct = round(tokens / budget * 100)
|
|
617
|
+
budget_k = f"{budget // 1000}k"
|
|
618
|
+
next_step = (
|
|
619
|
+
"Finish only the immediate step, then send the parent agent detailed findings, "
|
|
620
|
+
"completed work, and remaining work, and stop."
|
|
621
|
+
if tier >= 2
|
|
622
|
+
else "Keep working normally; at the next natural stopping point, send detailed progress "
|
|
623
|
+
"and any remaining work to the parent agent."
|
|
624
|
+
)
|
|
625
|
+
return (
|
|
626
|
+
f"[context-warning hook] This warning applies to a sub-agent thread, not the "
|
|
627
|
+
f"parent/main agent. This sub-agent is at {k} tokens - {pct}% of its {budget_k} "
|
|
628
|
+
f"{limit_name(priced)}. {next_step} Do not create a user-facing session handoff or "
|
|
629
|
+
f"claim that the main agent's context is full."
|
|
630
|
+
)
|
|
631
|
+
|
|
632
|
+
|
|
633
|
+
def subagent_reminder(tokens, budget, priced):
|
|
634
|
+
return (
|
|
635
|
+
f"[context-warning hook] Handoff mode is active for this sub-agent thread, not the "
|
|
636
|
+
f"parent/main agent: {headroom(tokens, budget)} left of its {budget // 1000}k "
|
|
637
|
+
f"{limit_name(priced)}. Keep findings compact and send them to the parent agent "
|
|
638
|
+
f"before the budget runs out. Do not create a user-facing session handoff."
|
|
639
|
+
)
|
|
640
|
+
|
|
641
|
+
|
|
642
|
+
def emit(event, additional_context, system_message=None):
|
|
643
|
+
payload = {
|
|
644
|
+
"hookSpecificOutput": {
|
|
645
|
+
"hookEventName": event,
|
|
646
|
+
"additionalContext": additional_context,
|
|
647
|
+
}
|
|
648
|
+
}
|
|
649
|
+
if system_message:
|
|
650
|
+
payload["systemMessage"] = system_message
|
|
651
|
+
payload["suppressOutput"] = True
|
|
652
|
+
print(json.dumps(payload))
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
def read_state(path):
|
|
656
|
+
"""{"tier": n, "stop_tier": n} from the state file; zeroes when absent or unreadable."""
|
|
657
|
+
try:
|
|
658
|
+
with open(path, encoding="utf-8") as f:
|
|
659
|
+
raw = f.read().strip()
|
|
660
|
+
except OSError:
|
|
661
|
+
return {"tier": 0, "stop_tier": 0}
|
|
662
|
+
try:
|
|
663
|
+
state = json.loads(raw)
|
|
664
|
+
if isinstance(state, dict):
|
|
665
|
+
return {"tier": int(state.get("tier", 0)), "stop_tier": int(state.get("stop_tier", 0))}
|
|
666
|
+
except (ValueError, TypeError):
|
|
667
|
+
pass
|
|
668
|
+
try:
|
|
669
|
+
return {"tier": int(raw or 0), "stop_tier": 0}
|
|
670
|
+
except ValueError:
|
|
671
|
+
return {"tier": 0, "stop_tier": 0}
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
def write_state(path, state):
|
|
675
|
+
try:
|
|
676
|
+
with open(path, "w", encoding="utf-8") as f:
|
|
677
|
+
json.dump(state, f)
|
|
678
|
+
except OSError:
|
|
679
|
+
pass
|
|
680
|
+
|
|
681
|
+
|
|
376
682
|
def main():
|
|
377
683
|
agent = agent_name(sys.argv[1:])
|
|
378
684
|
profile = AGENT_PROFILES.get(agent)
|
|
@@ -380,6 +686,7 @@ def main():
|
|
|
380
686
|
return
|
|
381
687
|
|
|
382
688
|
data = json.load(sys.stdin)
|
|
689
|
+
event = data.get("hook_event_name") or "UserPromptSubmit"
|
|
383
690
|
root = repo_root(data)
|
|
384
691
|
local_config = load_local_config(root)
|
|
385
692
|
if disabled_locally(local_config):
|
|
@@ -387,58 +694,89 @@ def main():
|
|
|
387
694
|
|
|
388
695
|
profile = resolve_profile(profile, handoff_skill(root))
|
|
389
696
|
|
|
697
|
+
# PostToolBatch and Stop also fire inside an Agent-tool sub-agent, where the transcript
|
|
698
|
+
# read here belongs to the parent. Telling a sub-agent to hand off the user's session
|
|
699
|
+
# would be wrong, and its own token count is not available, so stay silent.
|
|
700
|
+
if data.get("agent_id") and event in ("PostToolBatch", "Stop"):
|
|
701
|
+
return
|
|
702
|
+
|
|
390
703
|
transcript_path = data.get("transcript_path")
|
|
391
704
|
session_id = data.get("session_id", "unknown")
|
|
392
705
|
auto_mode = data.get("permission_mode") in profile["auto_modes"] and not auto_handoff_save_disabled(local_config)
|
|
393
|
-
|
|
706
|
+
has_transcript = bool(transcript_path) and os.path.isfile(transcript_path)
|
|
707
|
+
|
|
708
|
+
# SessionStart fires before the first request, so it must not need a transcript to exist.
|
|
709
|
+
if not has_transcript and event != "SessionStart":
|
|
394
710
|
return
|
|
395
711
|
|
|
396
|
-
|
|
712
|
+
if has_transcript:
|
|
713
|
+
tokens, model, reported_window, thread_id, is_subagent = CONTEXT_READERS[agent](transcript_path)
|
|
714
|
+
else:
|
|
715
|
+
tokens, model, reported_window, thread_id, is_subagent = 0, None, None, None, False
|
|
397
716
|
window = reported_window or profile["default_window"] or window_for(model)
|
|
398
717
|
boundary = premium_boundary_for(profile, model)
|
|
399
718
|
budget = min(window, boundary) if boundary else window
|
|
400
|
-
|
|
401
|
-
critical_tokens = int(budget * CRITICAL_FRACTION)
|
|
402
|
-
tier = 2 if tokens >= critical_tokens else 1 if tokens >= warn_tokens else 0
|
|
719
|
+
priced = budget < window
|
|
403
720
|
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
721
|
+
if event == "SessionStart":
|
|
722
|
+
emit(event, session_start_message(budget, window, model))
|
|
723
|
+
return
|
|
724
|
+
|
|
725
|
+
tier = 0
|
|
726
|
+
for level, fraction in ((3, FINAL_FRACTION), (2, CRITICAL_FRACTION), (1, WARN_FRACTION)):
|
|
727
|
+
if tokens >= int(budget * fraction):
|
|
728
|
+
tier = level
|
|
729
|
+
break
|
|
730
|
+
|
|
731
|
+
state_id = thread_id or session_id
|
|
732
|
+
safe_state_id = "".join(char for char in str(state_id) if char.isalnum() or char in "-_")[:128]
|
|
733
|
+
state_file = os.path.join(tempfile.gettempdir(), f"{agent}-context-warning-{safe_state_id or 'unknown'}")
|
|
734
|
+
state = read_state(state_file)
|
|
735
|
+
prev_tier = state["tier"]
|
|
736
|
+
|
|
737
|
+
if event == "Stop":
|
|
738
|
+
# Below the critical tier the mid-run notice is enough; interrupting the end of a turn
|
|
739
|
+
# at 70% costs more than it saves.
|
|
740
|
+
if tier < 2 or tier <= state["stop_tier"] or is_subagent:
|
|
741
|
+
return
|
|
742
|
+
write_state(state_file, {**state, "stop_tier": tier})
|
|
743
|
+
system_message, additional_context = stop_messages(profile, tier, tokens, budget, priced)
|
|
744
|
+
if profile["emits_system_message"]:
|
|
745
|
+
emit(event, additional_context, system_message)
|
|
746
|
+
else:
|
|
747
|
+
emit(event, additional_context)
|
|
748
|
+
return
|
|
413
749
|
|
|
414
750
|
if tier != prev_tier:
|
|
415
|
-
|
|
416
|
-
with open(state_file, "w", encoding="utf-8") as f:
|
|
417
|
-
f.write(str(tier))
|
|
418
|
-
except OSError:
|
|
419
|
-
pass
|
|
751
|
+
write_state(state_file, {**state, "tier": tier, "stop_tier": min(state["stop_tier"], tier)})
|
|
420
752
|
|
|
421
753
|
if tier <= prev_tier:
|
|
422
|
-
|
|
754
|
+
# Below the first threshold, or the context shrank - the state was re-armed above.
|
|
755
|
+
if tier == 0:
|
|
756
|
+
return
|
|
757
|
+
# Only a prompt repeats the reminder. On a tool batch it would be re-injected before
|
|
758
|
+
# every model request, which costs more context than it saves.
|
|
759
|
+
if event != "UserPromptSubmit":
|
|
760
|
+
return
|
|
761
|
+
emit(
|
|
762
|
+
event,
|
|
763
|
+
subagent_reminder(tokens, budget, priced)
|
|
764
|
+
if is_subagent
|
|
765
|
+
else reminder(tier, tokens, budget, priced),
|
|
766
|
+
)
|
|
767
|
+
return
|
|
768
|
+
|
|
769
|
+
if is_subagent:
|
|
770
|
+
emit(event, subagent_message(tier, tokens, budget, priced))
|
|
771
|
+
return
|
|
423
772
|
|
|
424
773
|
system_message, additional_context = messages(
|
|
425
|
-
profile, tier, tokens, budget,
|
|
774
|
+
profile, tier, tokens, budget, priced, auto_mode
|
|
426
775
|
)
|
|
427
|
-
|
|
428
|
-
if not profile["emits_system_message"]:
|
|
429
|
-
additional_context = f"{additional_context}\n\nTell the user: {system_message}"
|
|
430
|
-
|
|
431
|
-
payload = {
|
|
432
|
-
"hookSpecificOutput": {
|
|
433
|
-
"hookEventName": "UserPromptSubmit",
|
|
434
|
-
"additionalContext": additional_context,
|
|
435
|
-
}
|
|
436
|
-
}
|
|
437
776
|
if profile["emits_system_message"]:
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
print(json.dumps(payload))
|
|
777
|
+
emit(event, additional_context, system_message)
|
|
778
|
+
else:
|
|
779
|
+
emit(event, f"{additional_context}\n\nTell the user: {system_message}")
|
|
442
780
|
|
|
443
781
|
|
|
444
782
|
if __name__ == "__main__":
|