@ethlete/agent-rules 0.1.0-next.14 → 0.1.0-next.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/CHANGELOG.md +66 -0
  2. package/README.md +36 -2
  3. package/THIRD-PARTY-LICENSES.md +35 -0
  4. package/content/hooks/context-warning.py +450 -112
  5. package/content/hooks/subagent-model-policy.py +185 -0
  6. package/content/rules/subagent-models.md +29 -0
  7. package/content/skills/codex-subagent/SKILL.md +90 -0
  8. package/content/skills/codex-subagent/codex-agent.mjs +229 -0
  9. package/content/skills/design-exploration/SKILL.md +203 -0
  10. package/content/skills/design-exploration/check-story.mjs +139 -0
  11. package/content/skills/design-exploration/shoot-template.mjs +61 -0
  12. package/content/skills/domain-modeling/ADR-FORMAT.md +47 -0
  13. package/content/skills/domain-modeling/CONTEXT-FORMAT.md +60 -0
  14. package/content/skills/domain-modeling/SKILL.md +80 -0
  15. package/content/skills/grill-with-docs/SKILL.md +13 -0
  16. package/content/skills/grilling/SKILL.md +34 -0
  17. package/content/skills/handoff/SKILL.md +69 -4
  18. package/content/skills/sdk-update/SKILL.md +2 -1
  19. package/content/skills/timetrack/SKILL.md +190 -10
  20. package/package.json +1 -1
  21. package/src/index.js +2 -1
  22. package/src/index.js.map +1 -1
  23. package/src/lib/frontmatter.d.ts +2 -0
  24. package/src/lib/frontmatter.js +2 -1
  25. package/src/lib/frontmatter.js.map +1 -1
  26. package/src/lib/git-flow/config.js +1 -1
  27. package/src/lib/git-flow/config.js.map +1 -1
  28. package/src/lib/git.d.ts +37 -0
  29. package/src/lib/git.js +77 -1
  30. package/src/lib/git.js.map +1 -1
  31. package/src/lib/index.d.ts +1 -0
  32. package/src/lib/index.js +1 -0
  33. package/src/lib/index.js.map +1 -1
  34. package/src/lib/plain-text.d.ts +9 -0
  35. package/src/lib/plain-text.js +22 -0
  36. package/src/lib/plain-text.js.map +1 -0
  37. package/src/lib/targets/claude-hooks.js +2 -1
  38. package/src/lib/targets/claude-hooks.js.map +1 -1
  39. package/src/lib/targets/codex-hooks.js +2 -1
  40. package/src/lib/targets/codex-hooks.js.map +1 -1
  41. package/src/lib/targets/hooks-shared.d.ts +22 -1
  42. package/src/lib/targets/hooks-shared.js +42 -9
  43. package/src/lib/targets/hooks-shared.js.map +1 -1
  44. package/src/lib/targets/shared.js +1 -0
  45. package/src/lib/targets/shared.js.map +1 -1
  46. package/src/lib/timetrack-command.js +449 -20
  47. package/src/lib/timetrack-command.js.map +1 -1
  48. package/src/lib/timetrack.d.ts +241 -0
  49. package/src/lib/timetrack.js +83 -2
  50. package/src/lib/timetrack.js.map +1 -1
@@ -1,21 +1,40 @@
1
1
  #!/usr/bin/env python3
2
- """UserPromptSubmit hook: warn when the context window is getting large.
2
+ """Context-budget hook: keep the agent aware of how much context it may still spend.
3
3
 
4
4
  Reads the hook input JSON from stdin, estimates the current context size from
5
5
  the session transcript, and emits a warning (visible to both the user and the
6
6
  agent) when it crosses a threshold. Recommends the handoff skill so work can
7
7
  continue in a fresh session.
8
8
 
9
+ Registered on four events, because a warning the agent cannot see while it works
10
+ is not a warning:
11
+
12
+ * SessionStart - states the budget and how to scope work to it, before the
13
+ first request. Nothing else tells the model what its budget is.
14
+ * UserPromptSubmit - the tiered warning, plus a short reminder on every later
15
+ prompt.
16
+ * PostToolBatch - the same tiered warning, delivered mid-run. This is the only
17
+ event that fires between an agent's own model requests, so without it a long
18
+ autonomous run never learns its token count until the user types something.
19
+ * Stop - fires as the agent tries to end its turn. Its additionalContext is
20
+ delivered to the model and the conversation continues, so the decision to
21
+ finish or hand off happens without the user having to ask for it.
22
+
9
23
  Runs under both Claude Code and Codex, selected by `--agent` on the command
10
- line rather than sniffed from the payload — the generator writes the flag, so
11
- the two never have to be told apart at runtime. The two differ in three ways
12
- that matter here, all captured in AGENT_PROFILES:
24
+ line rather than sniffed from the payload - the generator writes the flag, so
25
+ the two never have to be told apart at runtime. Codex has only UserPromptSubmit;
26
+ the other three events are registered for Claude alone. The two differ in three
27
+ ways that matter here, all captured in AGENT_PROFILES:
13
28
 
14
29
  * Transcript format. Claude writes one JSON object per message with
15
30
  `message.usage`; Codex writes a rollout stream whose `token_count` events
16
31
  carry a TokenUsageInfo. Codex's rollout format is explicitly not a stable
17
32
  interface, so the parser searches each line for the usage object instead of
18
33
  walking a fixed path.
34
+ * Sub-agents. Codex gives each sub-agent its own rollout and thread id while
35
+ retaining the root session id. Warnings identify that thread explicitly and
36
+ tell it to report back to its parent instead of making the main agent hand
37
+ off its session.
19
38
  * Context budget. Claude's budget is capped at its 200k long-context pricing
20
39
  boundary. Codex uses the model-specific pricing boundary where OpenAI
21
40
  documents one, and the reported window otherwise. Crossing either boundary
@@ -23,7 +42,7 @@ that matter here, all captured in AGENT_PROFILES:
23
42
  * How a handoff is invoked. Claude has a slash command and /clear; Codex has
24
43
  neither and reads the skill from disk. The skill's own name differs per repo
25
44
  (`ethlete-handoff` where the generator installed it, `handoff` where the repo
26
- ships its own copy), so it is resolved at runtime — see handoff_skill().
45
+ ships its own copy), so it is resolved at runtime - see handoff_skill().
27
46
 
28
47
  At the critical tier, if the session's permission_mode is "auto", the
29
48
  instruction escalates from "recommend" to "just do it": auto mode already
@@ -34,8 +53,21 @@ handoff file itself immediately rather than waiting for a natural stopping
34
53
  point. Codex's permission_mode value set is undocumented, so its profile lists
35
54
  no auto modes and the escalation stays off there.
36
55
 
56
+ Neither the critical tier nor the Stop event forces a handoff. A session is
57
+ often a step or two from done when they fire, and some sessions hold reasoning
58
+ that no handoff file survives - crossing the boundary is then the cheaper
59
+ choice. So the Stop event offers three options and demands the agent name the
60
+ one it took: finish, hand off, or continue past the boundary for a reason it
61
+ states in one sentence. A third tier at 95% withdraws only the "finish" option,
62
+ because a budget that small covers no task.
63
+
37
64
  Warns once per tier per session (state kept in a temp file); re-arms itself
38
- if the context shrinks again (e.g. after a compaction).
65
+ if the context shrinks again (e.g. after a compaction). The Stop event keeps its
66
+ own per-tier counter, so crossing a tier mid-run costs at most one notice while
67
+ working and one demand at the end of the turn. Every prompt after the first
68
+ warning carries a short reminder instead: how much budget is left, and how to
69
+ work so that either ending stays cheap. One warning at 70% is stale by the time
70
+ it matters, and the agent cannot see its own token count.
39
71
 
40
72
  Can be disabled per machine via a gitignored ethlete-agents.config.local.json
41
73
  at the repo root: {"disableHooks": true} or {"disableHooks": ["context-warning"]}.
@@ -43,7 +75,8 @@ To keep the tiered warnings but drop just the auto-mode auto-save escalation,
43
75
  use {"disableAutoHandoffSave": true} instead - the critical tier then falls
44
76
  back to recommending a handoff, same as non-auto mode.
45
77
 
46
- Fail-safe: any error exits 0 with no output — the hook must never block a prompt.
78
+ Fail-safe: any error exits 0 with no output - the hook must never block a prompt,
79
+ a tool batch or the end of a turn.
47
80
  """
48
81
 
49
82
  import json
@@ -54,12 +87,38 @@ import tempfile
54
87
  HOOK_NAME = "context-warning"
55
88
  LOCAL_CONFIG_FILE = "ethlete-agents.config.local.json"
56
89
 
57
- # Warn / critical fire at these fractions of the token budget.
90
+ # Warn / critical / final fire at these fractions of the token budget.
58
91
  WARN_FRACTION = 0.70
59
92
  CRITICAL_FRACTION = 0.85
93
+ FINAL_FRACTION = 0.95
94
+
95
+ # How to work once the budget is tight - repeated on every prompt from the first
96
+ # warning on, so it stays in view as the budget shrinks.
97
+ HANDOFF_MODE = (
98
+ "Work in slices that each end in a committable state. Send bulky reads, searches and "
99
+ "test runs to a sub-agent so their output stays out of this context. Write down each "
100
+ "decision and dead end as you reach it, in the commit message or the handoff file. "
101
+ "Start nothing you cannot finish or hand off inside the remaining budget."
102
+ )
103
+
104
+ # Said once at session start, before anything has been read - the only point at which the
105
+ # scope of the whole session is still open.
106
+ SESSION_BUDGET = (
107
+ "Scope the session to it. Send bulky reads, searches and test runs to a sub-agent so "
108
+ "their output never enters this context. Commit in slices. A task that plainly does not "
109
+ "fit should be split now, not abandoned at 90%."
110
+ )
111
+
112
+ # The escape from a handoff at the critical tier, with the test that keeps it honest.
113
+ FINISH_FIRST = (
114
+ "If you can name every step that is left, and they plainly fit the remaining budget, "
115
+ "finish the task, commit, and then apply the handoff skill's own save test - with "
116
+ "everything committed and no step left, write no handoff file and say so instead. "
117
+ "\"Nearly done\" means you can name the last steps now, not that the end feels close."
118
+ )
60
119
 
61
120
  # On models with a window larger than this, tokens beyond it bill the whole
62
- # context at the long-context premium rate — so the budget never exceeds it.
121
+ # context at the long-context premium rate - so the budget never exceeds it.
63
122
  PREMIUM_BOUNDARY = 200_000
64
123
 
65
124
  # Codex models whose long-context pricing starts above 272k input tokens. Models
@@ -72,7 +131,7 @@ CODEX_PREMIUM_BOUNDARIES = (
72
131
  )
73
132
 
74
133
  # Context window (tokens) per model, matched by substring against the model id
75
- # from the transcript — first match wins. Edit these as model windows change;
134
+ # from the transcript - first match wins. Edit these as model windows change;
76
135
  # anything unmatched falls back to DEFAULT_WINDOW.
77
136
  CONTEXT_WINDOWS = (
78
137
  ("opus-5", 1_000_000),
@@ -80,7 +139,7 @@ CONTEXT_WINDOWS = (
80
139
  ("sonnet-4-5", 1_000_000),
81
140
  ("sonnet-5", 1_000_000),
82
141
  ("fable-5", 1_000_000),
83
- # Generic fallbacks — keep these last: the first substring match wins, so a bare
142
+ # Generic fallbacks - keep these last: the first substring match wins, so a bare
84
143
  # "opus"/"sonnet" entry placed above would shadow every versioned entry below it.
85
144
  ("opus", 200_000),
86
145
  ("sonnet", 200_000),
@@ -88,6 +147,13 @@ CONTEXT_WINDOWS = (
88
147
  )
89
148
  DEFAULT_WINDOW = 200_000
90
149
 
150
+ # Bytes read from the end of a Claude transcript to find the newest usage report. Read on
151
+ # every tool batch, so a full parse of a multi-megabyte transcript would add latency to
152
+ # every model request. Grows by GROWTH_FACTOR until a usage report is found.
153
+ TAIL_BYTES = 256 * 1024
154
+ MAX_TAIL_BYTES = 16 * 1024 * 1024
155
+ GROWTH_FACTOR = 8
156
+
91
157
  AGENT_PROFILES = {
92
158
  "claude": {
93
159
  "premium_boundary": PREMIUM_BOUNDARY,
@@ -101,11 +167,17 @@ AGENT_PROFILES = {
101
167
  "recommend": "recommend the user run /{handoff} to save state and start a fresh session",
102
168
  "suggest": "suggest the user run /{handoff} to save state and start a fresh session",
103
169
  "save_now": (
104
- "run the handoff skill's save mode right now (finish only an in-flight atomic "
105
- "edit first, nothing new). Then tell the user exactly which handoff file was "
106
- "written and that they should run /clear, then '/{handoff} resume <slug>', to "
107
- "continue — clearing and resuming can't be done programmatically, so this is "
108
- "the one step still on them."
170
+ "run the handoff skill's save mode right now - finish only an in-flight atomic "
171
+ "edit first, nothing new. Apply its own test: with every change committed, no "
172
+ "step left and nothing running, write no file, say that instead, and recommend "
173
+ "/clear. If a file is warranted, tell the user exactly which one was written and "
174
+ "that they should run /clear, then '/{handoff} resume <slug>', to continue - "
175
+ "clearing and resuming can't be done programmatically, so this is the one step "
176
+ "still on them."
177
+ ),
178
+ "save_at_stop": (
179
+ "save a handoff with the handoff skill, then tell the user which file was "
180
+ "written and that they should run /clear, then '/{handoff} resume <slug>'"
109
181
  ),
110
182
  },
111
183
  "codex": {
@@ -132,10 +204,15 @@ AGENT_PROFILES = {
132
204
  "continuing in a fresh session"
133
205
  ),
134
206
  "save_now": (
135
- "follow .agents/skills/{handoff}/SKILL.md and save a handoff right now "
136
- "(finish only an in-flight atomic edit first, nothing new). Then tell the user "
137
- "exactly which handoff file was written and that they should start a fresh "
138
- "codex session and resume from it."
207
+ "follow .agents/skills/{handoff}/SKILL.md and save a handoff right now - finish "
208
+ "only an in-flight atomic edit first, nothing new. Apply its own test: with "
209
+ "every change committed, no step left and nothing running, write no file and say "
210
+ "that instead. If a file is warranted, tell the user exactly which one was "
211
+ "written and that they should start a fresh codex session and resume from it."
212
+ ),
213
+ "save_at_stop": (
214
+ "save a handoff via .agents/skills/{handoff}/SKILL.md, then tell the user which "
215
+ "file was written and that a fresh session should resume from it"
139
216
  ),
140
217
  },
141
218
  }
@@ -145,7 +222,7 @@ def handoff_skill(root):
145
222
  """Name of the handoff skill installed in this repo.
146
223
 
147
224
  The generator prefixes every skill it writes, so the command is /ethlete-handoff in a
148
- consumer repo — but a repo that excludes the generated copy and ships its own (the SDK
225
+ consumer repo - but a repo that excludes the generated copy and ships its own (the SDK
149
226
  itself) has a plain /handoff instead. Naming the wrong one sends the user to a slash
150
227
  command that does not exist, so it is read off disk rather than assumed.
151
228
  """
@@ -165,7 +242,7 @@ def resolve_profile(profile, skill):
165
242
 
166
243
 
167
244
  def agent_name(argv):
168
- """The --agent value, defaulting to claude — older registrations pass no flag."""
245
+ """The --agent value, defaulting to claude - older registrations pass no flag."""
169
246
  for index, arg in enumerate(argv):
170
247
  if arg == "--agent" and index + 1 < len(argv):
171
248
  return argv[index + 1]
@@ -233,35 +310,62 @@ def premium_boundary_for(profile, model):
233
310
  return profile["premium_boundary"]
234
311
 
235
312
 
313
+ def tail_lines(path, limit):
314
+ """(lines newest-first, whole file read) for the last `limit` bytes of a file.
315
+
316
+ A partial first line is dropped, so every returned line is complete.
317
+ """
318
+ size = os.path.getsize(path)
319
+ start = max(size - limit, 0)
320
+ with open(path, "rb") as f:
321
+ f.seek(start)
322
+ chunk = f.read()
323
+ if start > 0:
324
+ newline = chunk.find(b"\n")
325
+ chunk = chunk[newline + 1 :] if newline != -1 else b""
326
+ return list(reversed(chunk.split(b"\n"))), start == 0
327
+
328
+
329
+ def claude_usage_in(lines):
330
+ """(usage, model) from the newest main-chain assistant message in `lines`, or (None, None)."""
331
+ for line in lines:
332
+ if not line.strip():
333
+ continue
334
+ try:
335
+ obj = json.loads(line)
336
+ except (json.JSONDecodeError, UnicodeDecodeError):
337
+ continue
338
+ if obj.get("type") != "assistant" or obj.get("isSidechain"):
339
+ continue
340
+ message = obj.get("message") or {}
341
+ usage = message.get("usage")
342
+ if usage:
343
+ return usage, message.get("model")
344
+ return None, None
345
+
346
+
236
347
  def claude_context_state(transcript_path):
237
- """(tokens, model, window) from the last main-chain assistant message.
348
+ """(tokens, model, window, thread id, is sub-agent) from the main-chain assistant message.
238
349
 
239
- tokens ≈ its total input tokens (fresh + cache read + cache creation). Claude reports no
240
- window in the transcript, so it is resolved from the model id by the caller.
350
+ tokens is its total input tokens (fresh + cache read + cache creation). Claude reports no
351
+ window in the transcript, so it is resolved from the model id by the caller. Only the tail
352
+ of the transcript is read, growing until a usage report is found - this runs on every tool
353
+ batch, and a transcript reaches tens of megabytes.
241
354
  """
242
- last_usage = None
243
- last_model = None
244
- with open(transcript_path, encoding="utf-8") as f:
245
- for line in f:
246
- try:
247
- obj = json.loads(line)
248
- except json.JSONDecodeError:
249
- continue
250
- if obj.get("type") != "assistant" or obj.get("isSidechain"):
251
- continue
252
- message = obj.get("message") or {}
253
- usage = message.get("usage")
254
- if usage:
255
- last_usage = usage
256
- last_model = message.get("model") or last_model
257
- if not last_usage:
258
- return 0, last_model, None
259
- tokens = (
260
- last_usage.get("input_tokens", 0)
261
- + last_usage.get("cache_read_input_tokens", 0)
262
- + last_usage.get("cache_creation_input_tokens", 0)
263
- )
264
- return tokens, last_model, None
355
+ limit = TAIL_BYTES
356
+ while True:
357
+ lines, complete = tail_lines(transcript_path, limit)
358
+ usage, model = claude_usage_in(lines)
359
+ if usage:
360
+ tokens = (
361
+ usage.get("input_tokens", 0)
362
+ + usage.get("cache_read_input_tokens", 0)
363
+ + usage.get("cache_creation_input_tokens", 0)
364
+ )
365
+ return tokens, model, None, None, False
366
+ if complete or limit >= MAX_TAIL_BYTES:
367
+ return 0, None, None, None, False
368
+ limit *= GROWTH_FACTOR
265
369
 
266
370
 
267
371
  def find_token_usage_info(node):
@@ -286,20 +390,29 @@ def find_token_usage_info(node):
286
390
 
287
391
 
288
392
  def codex_context_state(transcript_path):
289
- """(tokens, model, window) from the last token_count event in a Codex rollout.
393
+ """(tokens, model, window, thread id, is sub-agent) from a Codex rollout.
290
394
 
291
- tokens is the last request's `total_tokens` — Codex's own `tokens_in_context_window()`
395
+ tokens is the last request's `total_tokens` - Codex's own `tokens_in_context_window()`
292
396
  is exactly that field, and `last_token_usage` is replaced per request while
293
397
  `total_token_usage` accumulates across the whole session.
294
398
  """
295
399
  last_info = None
296
400
  last_model = None
401
+ thread_id = None
402
+ is_subagent = False
297
403
  with open(transcript_path, encoding="utf-8") as f:
298
404
  for line in f:
299
405
  try:
300
406
  obj = json.loads(line)
301
407
  except json.JSONDecodeError:
302
408
  continue
409
+ if obj.get("type") == "session_meta":
410
+ payload = obj.get("payload") or {}
411
+ source = payload.get("source")
412
+ thread_id = payload.get("id") or thread_id
413
+ is_subagent = payload.get("thread_source") == "subagent" or (
414
+ isinstance(source, dict) and isinstance(source.get("subagent"), dict)
415
+ )
303
416
  model = (obj.get("payload") or {}).get("model") if isinstance(obj.get("payload"), dict) else None
304
417
  if isinstance(model, str) and model:
305
418
  last_model = model
@@ -307,72 +420,265 @@ def codex_context_state(transcript_path):
307
420
  if info:
308
421
  last_info = info
309
422
  if not last_info:
310
- return 0, last_model, None
423
+ return 0, last_model, None, thread_id, is_subagent
311
424
  usage = last_info.get("last_token_usage") or {}
312
425
  window = last_info.get("model_context_window")
313
- return usage.get("total_tokens", 0), last_model, window if isinstance(window, int) else None
426
+ return (
427
+ usage.get("total_tokens", 0),
428
+ last_model,
429
+ window if isinstance(window, int) else None,
430
+ thread_id,
431
+ is_subagent,
432
+ )
314
433
 
315
434
 
316
435
  CONTEXT_READERS = {"claude": claude_context_state, "codex": codex_context_state}
317
436
 
318
437
 
438
+ def headroom(tokens, budget):
439
+ """How much budget is left, as a short human string."""
440
+ left = max(budget - tokens, 0)
441
+ return f"~{left // 1000}k tokens" if left >= 1000 else "under 1k tokens"
442
+
443
+
444
+ def limit_name(priced):
445
+ return "long-context pricing boundary" if priced else "context window"
446
+
447
+
448
+ def cost_of_crossing(priced):
449
+ """What the user pays for one more token past the budget."""
450
+ if priced:
451
+ return "every later request is billed at the premium rate"
452
+ return "auto-compact fires and the detail of this session is summarised away"
453
+
454
+
455
+ def token_count(tokens):
456
+ """A token count as a short label: 1M rather than 1000k."""
457
+ if tokens >= 1_000_000 and tokens % 1_000_000 == 0:
458
+ return f"{tokens // 1_000_000}M"
459
+ return f"{tokens // 1000}k"
460
+
461
+
462
+ def session_start_message(budget, window, model):
463
+ """The one line said before the first request, naming the budget and how to fit it.
464
+
465
+ A session that starts fresh has no assistant message yet, so the model - and with it the
466
+ window - is unknown. The budget is not: it is the pricing boundary either way.
467
+ """
468
+ budget_label = token_count(budget)
469
+ if not model:
470
+ frame = (
471
+ "past it a model with a larger window bills every request at the premium rate, and a "
472
+ f"{budget_label} model auto-compacts the session away"
473
+ )
474
+ elif budget < window:
475
+ frame = (
476
+ f"the long-context pricing boundary of this model's {token_count(window)} window; "
477
+ "past it every request is billed at the premium rate"
478
+ )
479
+ else:
480
+ frame = "this model's whole context window; past it auto-compact summarises the session away"
481
+ return (
482
+ f"[context-warning hook] Context budget for this session: {budget_label} tokens - {frame}. "
483
+ f"{SESSION_BUDGET} You will be warned at 70%, 85% and 95%."
484
+ )
485
+
486
+
319
487
  def messages(profile, tier, tokens, budget, priced, auto_mode):
320
488
  """(systemMessage, additionalContext) for a tier.
321
489
 
322
- priced: the budget is the pricing boundary, not the window — the reason to
490
+ priced: the budget is the pricing boundary, not the window - the reason to
323
491
  hand off is cost, not an imminent auto-compact.
324
- auto_mode: at the critical tier, escalates from "recommend a handoff" to
325
- "save it now" — see the module docstring for why.
492
+ auto_mode: from the critical tier on, escalates from "recommend a handoff"
493
+ to "save it now" - see the module docstring for why.
326
494
  """
327
495
  k = f"~{tokens // 1000}k"
328
496
  pct = round(tokens / budget * 100)
329
497
  budget_k = f"{budget // 1000}k"
498
+ left = headroom(tokens, budget)
330
499
 
331
500
  if priced:
332
- approach = "about to cross" if tier == 2 else "approaching"
333
- headline = f"Context is at {k} tokens — {approach} the {budget_k} long-context pricing boundary"
501
+ approach = "about to cross" if tier >= 2 else "approaching"
502
+ headline = f"Context is at {k} tokens - {approach} the {budget_k} long-context pricing boundary"
334
503
  detail = (
335
- f"The context is at {k} tokens — {pct}% of the {budget_k} long-context "
336
- f"pricing boundary."
504
+ f"The context is at {k} tokens - {pct}% of the {budget_k} long-context "
505
+ f"pricing boundary, {left} left."
337
506
  )
507
+ pressure = "Past it every request is billed at the premium rate. "
508
+ tail = ""
338
509
  else:
339
510
  headline = f"Context is at {k} tokens ({pct}% of the {budget_k} window)"
340
511
  detail = (
341
- f"The context window is at {k} tokens — {pct}% of this model's "
342
- f"{budget_k} window"
512
+ f"The context window is at {k} tokens - {pct}% of this model's "
513
+ f"{budget_k} window, {left} left."
343
514
  )
515
+ pressure = ""
516
+ tail = " - auto-compact is imminent" if tier >= 3 else " - auto-compact is close" if tier == 2 else ""
344
517
 
345
- if tier == 2:
346
- threshold = f" (critical, ≥{int(CRITICAL_FRACTION * 100)}%)"
347
- detail = detail if priced else f"{detail}{threshold}."
348
- pressure = (
349
- "Past it every request is billed at the premium rate. "
350
- if priced
351
- else ""
518
+ if tier >= 3:
519
+ gone = (
520
+ f"The finish-first choice from the last warning is gone: {left} does not "
521
+ f"cover another task."
352
522
  )
353
- tail = "" if priced else " — auto-compact is imminent"
354
523
  if auto_mode:
355
524
  return (
356
525
  f"🔴 {headline}{tail}. Auto mode is active: saving a handoff now.",
526
+ f"[context-warning hook] {detail} {pressure}{gone} Auto mode is active, so "
527
+ f"{profile['save_now']}",
528
+ )
529
+ return (
530
+ f"🔴 {headline}{tail}. {profile['act_now']}",
531
+ f"[context-warning hook] {detail} {pressure}{gone} Stop after the current tool "
532
+ f"call, then {profile['recommend']}.",
533
+ )
534
+
535
+ if tier == 2:
536
+ if auto_mode:
537
+ return (
538
+ f"🔴 {headline}{tail}. Auto mode is active: finishing the task, or saving a handoff.",
357
539
  f"[context-warning hook] {detail} {pressure}Auto mode is active, so don't "
358
- f"just recommend a handoff — {profile['save_now']}",
540
+ f"just recommend a handoff. Choose one of two, and tell the user which. "
541
+ f"{FINISH_FIRST} Otherwise {profile['save_now']}",
359
542
  )
360
543
  return (
361
544
  f"🔴 {headline}{tail}. {profile['act_now']}",
362
- f"[context-warning hook] {detail} {pressure}Finish only the immediate step, "
363
- f"then {profile['recommend']}. Do not start new sub-tasks.",
545
+ f"[context-warning hook] {detail} {pressure}Choose one of two, and tell the user "
546
+ f"which. {FINISH_FIRST} Otherwise finish only the immediate step, then "
547
+ f"{profile['recommend']}. Either way, start no new sub-tasks.",
364
548
  )
365
549
 
366
- threshold = f" (≥{int(WARN_FRACTION * 100)}%)"
367
- detail = detail if priced else f"{detail}{threshold}."
368
- pressure = "Past it every request is billed at the premium rate. " if priced else ""
369
550
  return (
370
551
  f"🟡 {headline}. At the next natural stopping point, {profile['act_later']}.",
371
- f"[context-warning hook] {detail} {pressure}When the current task reaches a "
372
- f"natural stopping point, {profile['suggest']}. Keep working normally until then.",
552
+ f"[context-warning hook] {detail} {pressure}Handoff mode is active from here on. "
553
+ f"{HANDOFF_MODE} Keep working; when the current task reaches a natural stopping "
554
+ f"point, {profile['suggest']}.",
555
+ )
556
+
557
+
558
+ def stop_messages(profile, tier, tokens, budget, priced):
559
+ """(systemMessage, additionalContext) for a turn that is about to end over budget.
560
+
561
+ The turn continues after this, so the agent acts on it without the user having to
562
+ send a message. It is the only point where the agent is between tasks and can still
563
+ spend tokens, so the choice is put here in full - including the option to cross the
564
+ boundary on purpose, which some sessions are worth.
565
+ """
566
+ k = f"~{tokens // 1000}k"
567
+ pct = round(tokens / budget * 100)
568
+ budget_k = f"{budget // 1000}k"
569
+ left = headroom(tokens, budget)
570
+ options = [
571
+ f"Hand off: {profile['save_at_stop']}.",
572
+ (
573
+ "Cross the boundary on purpose: take this only for a reason you can name in one "
574
+ "sentence right now, for example the task is one step from done, or this session "
575
+ f"holds reasoning a handoff file would lose. Say the reason, say that "
576
+ f"{cost_of_crossing(priced)}, and keep working."
577
+ ),
578
+ ]
579
+ if tier < 3:
580
+ options.insert(0, f"Finish the task now: {FINISH_FIRST}")
581
+ numbered = " ".join(f"{index}. {text}" for index, text in enumerate(options, start=1))
582
+ withdrawn = (
583
+ ""
584
+ if tier < 3
585
+ else f" The finish-first option is withdrawn at this tier: {left} does not cover another task."
586
+ )
587
+ return (
588
+ f"🔴 Turn ending at {k} tokens ({pct}% of the {budget_k} {limit_name(priced)}). "
589
+ f"Deciding whether to hand off.",
590
+ f"[context-warning hook] This turn is about to end at {k} tokens - {pct}% of the "
591
+ f"{budget_k} {limit_name(priced)}, {left} left. Past it {cost_of_crossing(priced)}."
592
+ f"{withdrawn} Do not end the turn without taking one of these, and tell the user "
593
+ f"which you took. {numbered}",
373
594
  )
374
595
 
375
596
 
597
+ def reminder(tier, tokens, budget, priced):
598
+ """The short note repeated on every prompt after a tier was already warned about.
599
+
600
+ A single warning at 70% is stale by the time the budget is actually tight, and the
601
+ agent has no way to read its own token count between prompts.
602
+ """
603
+ stop = (
604
+ " You were already told to hand off. Do not take on new work."
605
+ if tier >= 2
606
+ else ""
607
+ )
608
+ return (
609
+ f"[context-warning hook] Handoff mode is active: {headroom(tokens, budget)} left of "
610
+ f"the {budget // 1000}k {limit_name(priced)}. {HANDOFF_MODE}{stop}"
611
+ )
612
+
613
+
614
+ def subagent_message(tier, tokens, budget, priced):
615
+ k = f"~{tokens // 1000}k"
616
+ pct = round(tokens / budget * 100)
617
+ budget_k = f"{budget // 1000}k"
618
+ next_step = (
619
+ "Finish only the immediate step, then send the parent agent detailed findings, "
620
+ "completed work, and remaining work, and stop."
621
+ if tier >= 2
622
+ else "Keep working normally; at the next natural stopping point, send detailed progress "
623
+ "and any remaining work to the parent agent."
624
+ )
625
+ return (
626
+ f"[context-warning hook] This warning applies to a sub-agent thread, not the "
627
+ f"parent/main agent. This sub-agent is at {k} tokens - {pct}% of its {budget_k} "
628
+ f"{limit_name(priced)}. {next_step} Do not create a user-facing session handoff or "
629
+ f"claim that the main agent's context is full."
630
+ )
631
+
632
+
633
+ def subagent_reminder(tokens, budget, priced):
634
+ return (
635
+ f"[context-warning hook] Handoff mode is active for this sub-agent thread, not the "
636
+ f"parent/main agent: {headroom(tokens, budget)} left of its {budget // 1000}k "
637
+ f"{limit_name(priced)}. Keep findings compact and send them to the parent agent "
638
+ f"before the budget runs out. Do not create a user-facing session handoff."
639
+ )
640
+
641
+
642
+ def emit(event, additional_context, system_message=None):
643
+ payload = {
644
+ "hookSpecificOutput": {
645
+ "hookEventName": event,
646
+ "additionalContext": additional_context,
647
+ }
648
+ }
649
+ if system_message:
650
+ payload["systemMessage"] = system_message
651
+ payload["suppressOutput"] = True
652
+ print(json.dumps(payload))
653
+
654
+
655
+ def read_state(path):
656
+ """{"tier": n, "stop_tier": n} from the state file; zeroes when absent or unreadable."""
657
+ try:
658
+ with open(path, encoding="utf-8") as f:
659
+ raw = f.read().strip()
660
+ except OSError:
661
+ return {"tier": 0, "stop_tier": 0}
662
+ try:
663
+ state = json.loads(raw)
664
+ if isinstance(state, dict):
665
+ return {"tier": int(state.get("tier", 0)), "stop_tier": int(state.get("stop_tier", 0))}
666
+ except (ValueError, TypeError):
667
+ pass
668
+ try:
669
+ return {"tier": int(raw or 0), "stop_tier": 0}
670
+ except ValueError:
671
+ return {"tier": 0, "stop_tier": 0}
672
+
673
+
674
+ def write_state(path, state):
675
+ try:
676
+ with open(path, "w", encoding="utf-8") as f:
677
+ json.dump(state, f)
678
+ except OSError:
679
+ pass
680
+
681
+
376
682
  def main():
377
683
  agent = agent_name(sys.argv[1:])
378
684
  profile = AGENT_PROFILES.get(agent)
@@ -380,6 +686,7 @@ def main():
380
686
  return
381
687
 
382
688
  data = json.load(sys.stdin)
689
+ event = data.get("hook_event_name") or "UserPromptSubmit"
383
690
  root = repo_root(data)
384
691
  local_config = load_local_config(root)
385
692
  if disabled_locally(local_config):
@@ -387,58 +694,89 @@ def main():
387
694
 
388
695
  profile = resolve_profile(profile, handoff_skill(root))
389
696
 
697
+ # PostToolBatch and Stop also fire inside an Agent-tool sub-agent, where the transcript
698
+ # read here belongs to the parent. Telling a sub-agent to hand off the user's session
699
+ # would be wrong, and its own token count is not available, so stay silent.
700
+ if data.get("agent_id") and event in ("PostToolBatch", "Stop"):
701
+ return
702
+
390
703
  transcript_path = data.get("transcript_path")
391
704
  session_id = data.get("session_id", "unknown")
392
705
  auto_mode = data.get("permission_mode") in profile["auto_modes"] and not auto_handoff_save_disabled(local_config)
393
- if not transcript_path or not os.path.isfile(transcript_path):
706
+ has_transcript = bool(transcript_path) and os.path.isfile(transcript_path)
707
+
708
+ # SessionStart fires before the first request, so it must not need a transcript to exist.
709
+ if not has_transcript and event != "SessionStart":
394
710
  return
395
711
 
396
- tokens, model, reported_window = CONTEXT_READERS[agent](transcript_path)
712
+ if has_transcript:
713
+ tokens, model, reported_window, thread_id, is_subagent = CONTEXT_READERS[agent](transcript_path)
714
+ else:
715
+ tokens, model, reported_window, thread_id, is_subagent = 0, None, None, None, False
397
716
  window = reported_window or profile["default_window"] or window_for(model)
398
717
  boundary = premium_boundary_for(profile, model)
399
718
  budget = min(window, boundary) if boundary else window
400
- warn_tokens = int(budget * WARN_FRACTION)
401
- critical_tokens = int(budget * CRITICAL_FRACTION)
402
- tier = 2 if tokens >= critical_tokens else 1 if tokens >= warn_tokens else 0
719
+ priced = budget < window
403
720
 
404
- state_file = os.path.join(
405
- tempfile.gettempdir(), f"{agent}-context-warning-{session_id}"
406
- )
407
- prev_tier = 0
408
- try:
409
- with open(state_file, encoding="utf-8") as f:
410
- prev_tier = int(f.read().strip() or 0)
411
- except (OSError, ValueError):
412
- pass
721
+ if event == "SessionStart":
722
+ emit(event, session_start_message(budget, window, model))
723
+ return
724
+
725
+ tier = 0
726
+ for level, fraction in ((3, FINAL_FRACTION), (2, CRITICAL_FRACTION), (1, WARN_FRACTION)):
727
+ if tokens >= int(budget * fraction):
728
+ tier = level
729
+ break
730
+
731
+ state_id = thread_id or session_id
732
+ safe_state_id = "".join(char for char in str(state_id) if char.isalnum() or char in "-_")[:128]
733
+ state_file = os.path.join(tempfile.gettempdir(), f"{agent}-context-warning-{safe_state_id or 'unknown'}")
734
+ state = read_state(state_file)
735
+ prev_tier = state["tier"]
736
+
737
+ if event == "Stop":
738
+ # Below the critical tier the mid-run notice is enough; interrupting the end of a turn
739
+ # at 70% costs more than it saves.
740
+ if tier < 2 or tier <= state["stop_tier"] or is_subagent:
741
+ return
742
+ write_state(state_file, {**state, "stop_tier": tier})
743
+ system_message, additional_context = stop_messages(profile, tier, tokens, budget, priced)
744
+ if profile["emits_system_message"]:
745
+ emit(event, additional_context, system_message)
746
+ else:
747
+ emit(event, additional_context)
748
+ return
413
749
 
414
750
  if tier != prev_tier:
415
- try:
416
- with open(state_file, "w", encoding="utf-8") as f:
417
- f.write(str(tier))
418
- except OSError:
419
- pass
751
+ write_state(state_file, {**state, "tier": tier, "stop_tier": min(state["stop_tier"], tier)})
420
752
 
421
753
  if tier <= prev_tier:
422
- return # already warned at this tier (or context shrank — state re-armed above)
754
+ # Below the first threshold, or the context shrank - the state was re-armed above.
755
+ if tier == 0:
756
+ return
757
+ # Only a prompt repeats the reminder. On a tool batch it would be re-injected before
758
+ # every model request, which costs more context than it saves.
759
+ if event != "UserPromptSubmit":
760
+ return
761
+ emit(
762
+ event,
763
+ subagent_reminder(tokens, budget, priced)
764
+ if is_subagent
765
+ else reminder(tier, tokens, budget, priced),
766
+ )
767
+ return
768
+
769
+ if is_subagent:
770
+ emit(event, subagent_message(tier, tokens, budget, priced))
771
+ return
423
772
 
424
773
  system_message, additional_context = messages(
425
- profile, tier, tokens, budget, budget < window, auto_mode
774
+ profile, tier, tokens, budget, priced, auto_mode
426
775
  )
427
-
428
- if not profile["emits_system_message"]:
429
- additional_context = f"{additional_context}\n\nTell the user: {system_message}"
430
-
431
- payload = {
432
- "hookSpecificOutput": {
433
- "hookEventName": "UserPromptSubmit",
434
- "additionalContext": additional_context,
435
- }
436
- }
437
776
  if profile["emits_system_message"]:
438
- payload["systemMessage"] = system_message
439
- payload["suppressOutput"] = True
440
-
441
- print(json.dumps(payload))
777
+ emit(event, additional_context, system_message)
778
+ else:
779
+ emit(event, f"{additional_context}\n\nTell the user: {system_message}")
442
780
 
443
781
 
444
782
  if __name__ == "__main__":