agent-usage-manager 0.2.4__tar.gz → 0.2.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_usage_manager-0.2.6/.github/workflows/publish.yml +41 -0
- agent_usage_manager-0.2.6/AGENTS.md +58 -0
- agent_usage_manager-0.2.6/CLAUDE.md +1 -0
- agent_usage_manager-0.2.4/README.md → agent_usage_manager-0.2.6/PKG-INFO +145 -7
- agent_usage_manager-0.2.4/PKG-INFO → agent_usage_manager-0.2.6/README.md +110 -26
- agent_usage_manager-0.2.4/agent_usage_manager/agents.yaml → agent_usage_manager-0.2.6/agent_usage_manager/agents.default.yaml +47 -26
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/agent_usage_manager/app.py +193 -22
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/agent_usage_manager/cli.py +87 -2
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/agent_usage_manager/static/index.html +98 -38
- agent_usage_manager-0.2.6/demo.tape +23 -0
- agent_usage_manager-0.2.6/docs/dashboard-live.gif +0 -0
- agent_usage_manager-0.2.6/docs/dashboard-live.png +0 -0
- agent_usage_manager-0.2.6/docs/demo.gif +0 -0
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/docs/design/HLD.md +40 -35
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/docs/design/LLD.md +183 -102
- agent_usage_manager-0.2.6/pyproject.toml +55 -0
- agent_usage_manager-0.2.6/tests/test_design_docs.py +299 -0
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/tests/test_smoke.py +185 -2
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/tests/test_synthetic.py +117 -0
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/uv.lock +1 -1
- agent_usage_manager-0.2.4/demo.tape +0 -21
- agent_usage_manager-0.2.4/pyproject.toml +0 -31
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/.github/workflows/ci.yml +0 -0
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/.gitignore +0 -0
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/LICENSE +0 -0
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/agent_usage_manager/__init__.py +0 -0
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/docs/dashboard.png +0 -0
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/requirements.txt +0 -0
- {agent_usage_manager-0.2.4 → agent_usage_manager-0.2.6}/run.sh +0 -0
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
|
|
7
|
+
permissions:
|
|
8
|
+
contents: read
|
|
9
|
+
|
|
10
|
+
jobs:
|
|
11
|
+
build:
|
|
12
|
+
runs-on: ubuntu-latest
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v4
|
|
15
|
+
- uses: actions/setup-python@v5
|
|
16
|
+
with:
|
|
17
|
+
python-version: "3.12"
|
|
18
|
+
- run: python -m pip install --upgrade pip build
|
|
19
|
+
- name: Verify tag matches pyproject version
|
|
20
|
+
run: |
|
|
21
|
+
PKG=$(python -c "import tomllib;print(tomllib.load(open('pyproject.toml','rb'))['project']['version'])")
|
|
22
|
+
TAG="${GITHUB_REF_NAME#v}"
|
|
23
|
+
[ "$PKG" = "$TAG" ] || { echo "tag $GITHUB_REF_NAME != pyproject version $PKG"; exit 1; }
|
|
24
|
+
- run: python -m build
|
|
25
|
+
- uses: actions/upload-artifact@v4
|
|
26
|
+
with:
|
|
27
|
+
name: dist
|
|
28
|
+
path: dist/
|
|
29
|
+
|
|
30
|
+
publish:
|
|
31
|
+
needs: build
|
|
32
|
+
runs-on: ubuntu-latest
|
|
33
|
+
environment: pypi
|
|
34
|
+
permissions:
|
|
35
|
+
id-token: write # trusted publishing (OIDC) — no API token stored anywhere
|
|
36
|
+
steps:
|
|
37
|
+
- uses: actions/download-artifact@v4
|
|
38
|
+
with:
|
|
39
|
+
name: dist
|
|
40
|
+
path: dist/
|
|
41
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# agent-usage-manager
|
|
2
|
+
|
|
3
|
+
## Design docs (HLD/LLD)
|
|
4
|
+
|
|
5
|
+
`docs/design/HLD.md` and `docs/design/LLD.md` are the architecture contract for this
|
|
6
|
+
repo. **Where a doc disagrees with the code, the code wins and the doc is stale** — that
|
|
7
|
+
is a bug to fix, not a tie to settle in the doc's favor.
|
|
8
|
+
|
|
9
|
+
A change to mapped code updates the owning doc **in the same PR**. Tiered, so routine
|
|
10
|
+
work does not churn the docs:
|
|
11
|
+
|
|
12
|
+
1. **Update the LLD** when a file, function signature, data model, schema, event shape,
|
|
13
|
+
or config/env surface **the LLD names** changes — or when a module appears or
|
|
14
|
+
disappears under an owned path.
|
|
15
|
+
2. **Update the HLD** only on a boundary change: a component added or retired, a new
|
|
16
|
+
port/service/external dependency, a changed trust boundary or auth gate, a changed
|
|
17
|
+
contract between components, or a subsystem promoted, demoted, or absorbed.
|
|
18
|
+
3. **Update neither** for bugfixes, styling, tests, or refactors that preserve every
|
|
19
|
+
surface the docs name.
|
|
20
|
+
|
|
21
|
+
Do not regenerate a doc wholesale to satisfy this. The `design-doc` skill **overwrites**
|
|
22
|
+
and will destroy hand-written intent (why a gate exists, what the design refuses to do);
|
|
23
|
+
use it to create a missing pair, never to refresh a live one. Edit the affected sections
|
|
24
|
+
and move the `**Refreshed:**` line.
|
|
25
|
+
|
|
26
|
+
`CLAUDE.md` is a **symlink to `AGENTS.md`** — edit `AGENTS.md`, never the symlink.
|
|
27
|
+
Claude Code prefers `CLAUDE.md` and ignores a real `AGENTS.md` when both exist, so the
|
|
28
|
+
symlink keeps one source of truth and stops anything that later writes a `CLAUDE.md`
|
|
29
|
+
from silently shadowing this contract. Codex and Kimi read `AGENTS.md` directly. The
|
|
30
|
+
antigravity lane reads no project file at all — give it self-contained prompts.
|
|
31
|
+
|
|
32
|
+
Ownership map — machine-readable, parsed by `tests/test_design_docs.py`. Add a row when
|
|
33
|
+
you add a package; `none` means "no design contract, deliberately". A dir with no row but
|
|
34
|
+
rows beneath it is a container and is recursed into, so a package dropped inside one
|
|
35
|
+
cannot inherit its parent's coverage.
|
|
36
|
+
|
|
37
|
+
```design-doc-map
|
|
38
|
+
agent_usage_manager/ -> docs/design/HLD.md docs/design/LLD.md
|
|
39
|
+
docs/ -> none
|
|
40
|
+
tests/ -> none
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
`python3 tests/test_design_docs.py` reports drift: per doc, the modules added or removed
|
|
44
|
+
under its owned paths since that doc last changed (test files excluded — a new test needs
|
|
45
|
+
no LLD entry). Added/removed modules are the signal; raw commit counts are noise and print
|
|
46
|
+
as context only. The pytest asserts the map is structurally sound; it deliberately does
|
|
47
|
+
**not** fail on drift, because a doc gate that blocks merges buys rubber-stamp edits,
|
|
48
|
+
not maintained docs. A file *modified* to change its public API will not flag — boundary
|
|
49
|
+
changes still need a human read.
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
`python3 tests/test_design_docs.py --audit` answers a different question: which tracked modules
|
|
53
|
+
are named in **no** design doc at all. The drift report only diffs forward from a doc's
|
|
54
|
+
own last change, so staleness that predates an incomplete refresh is invisible to it
|
|
55
|
+
permanently — paws described a `scene.js` renderer for three weeks after it became
|
|
56
|
+
`wool.js`, and no number of drift runs could have said so. Run `--audit` when you inherit
|
|
57
|
+
a doc you did not write. It skips tests, `__init__.py`, vendored code, type shims,
|
|
58
|
+
generated migration revisions, and anything under a `-> none` path.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
AGENTS.md
|
|
@@ -1,17 +1,52 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: agent-usage-manager
|
|
3
|
+
Version: 0.2.6
|
|
4
|
+
Summary: htop for AI agents — liveness, CPU/mem/GPU usage, and a kill switch for headless agents (openclaw, hermes, ollama, vllm, claude-code).
|
|
5
|
+
Project-URL: Homepage, https://github.com/minglong51/agent-usage-manager
|
|
6
|
+
Project-URL: Repository, https://github.com/minglong51/agent-usage-manager
|
|
7
|
+
Project-URL: Issues, https://github.com/minglong51/agent-usage-manager/issues
|
|
8
|
+
Project-URL: Newsletter, https://buttondown.com/minglong51
|
|
9
|
+
License: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: ai-agents,gpu,llm,monitoring,observability,ollama,vllm
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Environment :: Web Environment
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: System Administrators
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Operating System :: MacOS :: MacOS X
|
|
18
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
24
|
+
Classifier: Topic :: System :: Monitoring
|
|
25
|
+
Classifier: Topic :: System :: Systems Administration
|
|
26
|
+
Requires-Python: >=3.9
|
|
27
|
+
Requires-Dist: fastapi>=0.110
|
|
28
|
+
Requires-Dist: psutil>=5.9
|
|
29
|
+
Requires-Dist: pyyaml>=6.0
|
|
30
|
+
Requires-Dist: uvicorn[standard]>=0.27
|
|
31
|
+
Provides-Extra: dev
|
|
32
|
+
Requires-Dist: httpx>=0.27; extra == 'dev'
|
|
33
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
34
|
+
Description-Content-Type: text/markdown
|
|
35
|
+
|
|
1
36
|
# agent-usage-manager
|
|
2
37
|
|
|
3
38
|
A small web dashboard for **headless AI agents** running on a machine —
|
|
4
39
|
OpenClaw, Hermes, Claude Code, Ollama, vLLM, llama.cpp, or anything you name. It
|
|
5
40
|
shows which agents are alive and what they're costing you (CPU, memory, GPU), and
|
|
6
41
|
gives you a **kill button** per agent. Think `htop`, scoped to just your agents —
|
|
7
|
-
the [screenshot below](docs/dashboard.png) is a real run on a fleet node.
|
|
42
|
+
the [screenshot below](https://raw.githubusercontent.com/minglong51/agent-usage-manager/main/docs/dashboard.png) is a real run on a fleet node.
|
|
8
43
|
|
|
9
44
|
No database, no auth framework (one static token file gates the kill switch),
|
|
10
45
|
four direct dependencies (FastAPI, uvicorn, psutil, PyYAML). Runs on macOS and Linux. It is a
|
|
11
46
|
per-node monitor and guarded local control panel: fleet schedulers may consume
|
|
12
47
|
its read-only telemetry, but should own their own scheduling and actuation.
|
|
13
48
|
|
|
14
|
-

|
|
49
|
+

|
|
15
50
|
|
|
16
51
|
*A real run: ten agents grouped by process tree (`+N` = children rolled up),
|
|
17
52
|
per-agent CPU/memory/uptime, launchd-supervised jobs flagged, and a kill button
|
|
@@ -30,6 +65,20 @@ ollama 50122 ● running 3.1 9210 14080 6h 02m ollama ru
|
|
|
30
65
|
> run on Apple Silicon, where per-process GPU stats aren't available so that column
|
|
31
66
|
> is hidden. The UI auto-refreshes every 3s.
|
|
32
67
|
|
|
68
|
+
## Quick start
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
uvx agent-usage-manager # then open http://127.0.0.1:8765 (opens automatically)
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
One command — no install, no virtualenv, no leftovers. Other install options,
|
|
75
|
+
config, and flags: [Install & run](#install--run).
|
|
76
|
+
|
|
77
|
+

|
|
78
|
+
|
|
79
|
+
*Field notes on running agents with discipline go out on the
|
|
80
|
+
[Agent Discipline](https://buttondown.com/minglong51) list — a few emails a month.*
|
|
81
|
+
|
|
33
82
|
## What it does
|
|
34
83
|
|
|
35
84
|
- **One row per agent.** Agents are grouped by process tree — the spawned children
|
|
@@ -45,7 +94,9 @@ ollama 50122 ● running 3.1 9210 14080 6h 02m ollama ru
|
|
|
45
94
|
- **Trends, not just snapshots.** Each row has a CPU sparkline (last ~20 min, sampled
|
|
46
95
|
in the background even with no browser open), plus a **`hot 5m+`** badge when an
|
|
47
96
|
agent has been pegged ≥90% CPU for 5+ minutes, an **`idle 10m+`** badge when a
|
|
48
|
-
long-running agent has done nothing for 10+ minutes
|
|
97
|
+
long-running agent has done nothing for 10+ minutes (suppressed for labels in
|
|
98
|
+
`idle_ok:` — for a fleet that waits for work, idle is the normal state and
|
|
99
|
+
badging it is wallpaper), and a **`churn ×N`** badge when
|
|
49
100
|
the same agent has died young 3+ times in 10 minutes — the states worth investigating
|
|
50
101
|
(runaway, possibly wedged, crash-looping under a supervisor). Churn is what hot/idle
|
|
51
102
|
can't see: a crash-looping process is a fresh pid every poll, so no per-process
|
|
@@ -54,9 +105,15 @@ ollama 50122 ● running 3.1 9210 14080 6h 02m ollama ru
|
|
|
54
105
|
- **Alerts.** A dashboard only helps while you're looking at it. Add an `alerts:`
|
|
55
106
|
block to `agents.yaml` and any badge appearing runs your command (desktop
|
|
56
107
|
notification, Telegram bot, pager — anything) with the details in `$AUM_*` env
|
|
57
|
-
vars.
|
|
108
|
+
vars. `$AUM_MSG` leads with the plain-English verdict ("Codex is crash-looping —
|
|
109
|
+
4 restarts in 10 min"); the machine snapshot trails in brackets. Fires once per
|
|
110
|
+
transition with a cooldown, never from the `list` CLI, and the cooldown is only
|
|
111
|
+
charged when the command exits 0 (a broken notifier doesn't suppress the retry).
|
|
58
112
|
By default only `hot`/`churn`/`leak` alert — `idle` is the normal state of an
|
|
59
113
|
agent fleet that waits for work, so it's opt-in.
|
|
114
|
+
Run `agent-usage-manager test-alert` once after wiring it up: it fires the
|
|
115
|
+
configured command synchronously with a test message and reports the exit
|
|
116
|
+
status, so a broken channel surfaces today instead of during the next incident.
|
|
60
117
|
|
|
61
118
|
```yaml
|
|
62
119
|
alerts:
|
|
@@ -79,11 +136,25 @@ ollama 50122 ● running 3.1 9210 14080 6h 02m ollama ru
|
|
|
79
136
|
- **Config hot-reload.** Edits to `agents.yaml` apply on the next poll, no restart.
|
|
80
137
|
A broken edit keeps the last good config and shows the parse error in the header.
|
|
81
138
|
- **`list` subcommand.** `agent-usage-manager list` (or `list --json`) prints a one-shot
|
|
82
|
-
table to stdout — no server, good for scripts and cron checks.
|
|
139
|
+
table to stdout — no server, good for scripts and cron checks. One-shot mode has no
|
|
140
|
+
history, so the sustained-state flags (`hot`/`idle`/`churn`/`leak`) can never populate
|
|
141
|
+
there (`--json` says `"flags_available": false`); query the running server's
|
|
142
|
+
`/api/agents` when you need flags.
|
|
83
143
|
- **Kill-safe table.** Rows keep a stable order (sorted by label) and never reorder
|
|
84
144
|
while your pointer is over the table, so the kill button can't shift under your
|
|
85
145
|
cursor mid-click.
|
|
86
146
|
|
|
147
|
+
## Why not just htop / Grafana?
|
|
148
|
+
|
|
149
|
+
- **htop sees processes, not agents.** No rollup of a spawned tree under the
|
|
150
|
+
agent that owns it, no agent states (crash-looping, idle, leaking), and an
|
|
151
|
+
unguarded `F9`.
|
|
152
|
+
- **Grafana + Prometheus + an exporter is a stack you operate.** This is one
|
|
153
|
+
command with no database — and it still serves `/metrics` if you want both.
|
|
154
|
+
- **The kill switch is the point.** Allowlist-matched, token-gated, guarded
|
|
155
|
+
against CSRF and DNS rebinding, with an append-only action log — a web page
|
|
156
|
+
that can stop processes needs exactly those.
|
|
157
|
+
|
|
87
158
|
## Product boundary
|
|
88
159
|
|
|
89
160
|
`agent-usage-manager` is intentionally **not** a fleet scheduler, dispatcher, or
|
|
@@ -139,6 +210,25 @@ This is the important part — a web page that can kill processes needs guardrai
|
|
|
139
210
|
also pass `--unsafe-expose`. Exposing the port means one static token is all
|
|
140
211
|
that stands between the network and your agents — put real auth in front
|
|
141
212
|
(reverse proxy + basic auth, SSH tunnel, etc.) before using that flag.
|
|
213
|
+
- **A proxy voids the loopback guarantee.** The production deployment (Ming's
|
|
214
|
+
suite) fronts this loopback bind with `tailscale serve` (`:8448 → 127.0.0.1:8765`),
|
|
215
|
+
so the tailnet reaches the UI despite the loopback bind: reads like
|
|
216
|
+
`/api/agents` are open there, and the static token is the *only* boundary on
|
|
217
|
+
actions. That is deliberate (owner-only tailnet + token-gated actions), but do
|
|
218
|
+
not read "loopback-only" as "not network-reachable" — a proxy in front is not
|
|
219
|
+
a trust boundary.
|
|
220
|
+
- **Fronting it with a proxy needs `AUM_TRUSTED_HOSTS`.** The DNS-rebinding guard
|
|
221
|
+
refuses any request whose `Host` is a non-local DNS name — which is exactly what
|
|
222
|
+
a proxy forwards. Opt that one name in explicitly:
|
|
223
|
+
|
|
224
|
+
```bash
|
|
225
|
+
AUM_TRUSTED_HOSTS=box.your-tailnet.ts.net agent-usage-manager
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
Comma-separated for several. Only add a name nobody else can mint a cert for on
|
|
229
|
+
this node; a name you don't control reopens the hole the guard exists to close.
|
|
230
|
+
Empty by default — out of the box only `localhost`, `127.0.0.1` and `::1` are
|
|
231
|
+
accepted.
|
|
142
232
|
|
|
143
233
|
## Limits & known issues
|
|
144
234
|
|
|
@@ -170,7 +260,7 @@ This is the important part — a web page that can kill processes needs guardrai
|
|
|
170
260
|
- **Windows is untested.** Kill maps to `TerminateProcess` via psutil and may
|
|
171
261
|
work, but CI covers Linux + macOS only.
|
|
172
262
|
|
|
173
|
-
##
|
|
263
|
+
## Install & run
|
|
174
264
|
|
|
175
265
|
**Recommended — one command, nothing to install first:**
|
|
176
266
|
|
|
@@ -262,6 +352,19 @@ and `/metrics` series all use the derived label, so each instance gets its own
|
|
|
262
352
|
state. Sessions that don't match the regex keep their `agents:` label, and the
|
|
263
353
|
key is ignored where tmux isn't installed or running.
|
|
264
354
|
|
|
355
|
+
**Supervised fleets (`launchd_labels:`, macOS)** — the same problem for agents
|
|
356
|
+
that run as launchd jobs and never touch tmux (five `hermes` LaunchAgents all
|
|
357
|
+
landing as "hermes"). The launchd job label is their durable identity:
|
|
358
|
+
|
|
359
|
+
```yaml
|
|
360
|
+
launchd_labels: "^ai\\.hermes\\.(?:gateway-)?(.+)$" # ai.hermes.gateway-frontdoor → frontdoor
|
|
361
|
+
```
|
|
362
|
+
|
|
363
|
+
When a matched root's own launchd job label matches, the first capture group
|
|
364
|
+
(the whole label if there's no group) becomes the row label — with the same
|
|
365
|
+
per-instance churn/alert/metrics identity as `tmux_labels:`. `tmux_labels`
|
|
366
|
+
wins when both apply; roots with no matching job keep their `agents:` label.
|
|
367
|
+
|
|
265
368
|
**Which `agents.yaml` is used** — resolved once at startup, first hit wins:
|
|
266
369
|
|
|
267
370
|
1. `AGENTS_CONFIG=/path/to/agents.yaml` env var (the `--config` flag sets this)
|
|
@@ -330,7 +433,7 @@ Linux (systemd), `~/.config/systemd/user/agent-usage-manager.service`:
|
|
|
330
433
|
[Unit]
|
|
331
434
|
Description=agent usage manager
|
|
332
435
|
[Service]
|
|
333
|
-
ExecStart=%h/agent-usage-manager/.venv/bin/uvicorn app:app --port 8765
|
|
436
|
+
ExecStart=%h/agent-usage-manager/.venv/bin/uvicorn agent_usage_manager.app:app --port 8765
|
|
334
437
|
WorkingDirectory=%h/agent-usage-manager
|
|
335
438
|
Restart=on-failure
|
|
336
439
|
[Install]
|
|
@@ -355,6 +458,35 @@ SIGTERM/SIGKILL on POSIX and TerminateProcess on Windows.
|
|
|
355
458
|
|
|
356
459
|
## Release notes
|
|
357
460
|
|
|
461
|
+
### 0.2.5 — per-instance labels for supervised fleets + the honesty batch
|
|
462
|
+
|
|
463
|
+
- `launchd_labels:` config — per-instance row labels from launchd job labels,
|
|
464
|
+
mirroring `tmux_labels:`. A supervised fleet (several LaunchAgents on one
|
|
465
|
+
binary) no longer collapses into one blurred label for churn tracking, alert
|
|
466
|
+
transitions, and `/metrics`. `idle_ok:` suppression matches the base matcher
|
|
467
|
+
label as well, so a renamed instance (`hermes` → `frontdoor`) keeps its
|
|
468
|
+
class-level suppression.
|
|
469
|
+
- New `test-alert` subcommand: fires the configured `alerts.command` once,
|
|
470
|
+
synchronously, and reports the exit status — proves the alert channel before
|
|
471
|
+
an incident depends on it.
|
|
472
|
+
- `ignore:` now covers ChatGPT.app's embedded Codex helpers (renderer/service
|
|
473
|
+
processes, `Resources/codex` app-server) and the `Codex Computer Use`
|
|
474
|
+
desktop-automation app — they matched the `codex` pattern via the bundle
|
|
475
|
+
name and cluttered the dashboard with GUI plumbing.
|
|
476
|
+
- Config deletion is no longer silent: a vanished `agents.yaml` surfaces as a
|
|
477
|
+
`config_error` ("running on the last good config") instead of looking like
|
|
478
|
+
hot-reload still works.
|
|
479
|
+
- The kill confirm now names the agent (label + command line, not just PID)
|
|
480
|
+
and discloses that plain kill escalates SIGTERM → SIGKILL after 3s.
|
|
481
|
+
- The first-kill token prompt names the server's host (and the ssh one-liner)
|
|
482
|
+
for browsers viewing the dashboard over a tunnel.
|
|
483
|
+
- The DNS-rebinding 403 now names the remedy (use a loopback name or bare IP).
|
|
484
|
+
- `list` says so when flag fields can't populate (one-shot mode has no
|
|
485
|
+
history): a stderr note, and `"flags_available": false` in `--json`.
|
|
486
|
+
- `--host ::1` now produces a valid bracketed IPv6 URL for the browser open.
|
|
487
|
+
- Fixed the README's systemd unit (`uvicorn app:app` could never resolve the
|
|
488
|
+
module; it's `agent_usage_manager.app:app`).
|
|
489
|
+
|
|
358
490
|
### 0.2.4 — stop reporting non-agents
|
|
359
491
|
|
|
360
492
|
- `ignore:` now covers Sparkle's `Autoupdate`/`Updater` (Codex.app's equivalent of
|
|
@@ -417,6 +549,12 @@ SIGTERM/SIGKILL on POSIX and TerminateProcess on Windows.
|
|
|
417
549
|
closed; add `--unsafe-expose` only with auth in front (see
|
|
418
550
|
[Safety](#safety)).
|
|
419
551
|
|
|
552
|
+
## Stay in the loop
|
|
553
|
+
|
|
554
|
+
New tools and field notes on running AI agents with discipline go to the
|
|
555
|
+
[Agent Discipline](https://buttondown.com/minglong51) list first — launch
|
|
556
|
+
notes, operational patterns, early access. A few emails a month at most.
|
|
557
|
+
|
|
420
558
|
## License
|
|
421
559
|
|
|
422
560
|
MIT
|
|
@@ -1,36 +1,17 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: agent-usage-manager
|
|
3
|
-
Version: 0.2.4
|
|
4
|
-
Summary: htop for AI agents — liveness, CPU/mem/GPU usage, and a kill switch for headless agents (openclaw, hermes, ollama, vllm, claude-code).
|
|
5
|
-
Project-URL: Homepage, https://github.com/minglong51/agent-usage-manager
|
|
6
|
-
Project-URL: Repository, https://github.com/minglong51/agent-usage-manager
|
|
7
|
-
License: MIT
|
|
8
|
-
License-File: LICENSE
|
|
9
|
-
Keywords: ai-agents,gpu,llm,monitoring,observability,ollama,vllm
|
|
10
|
-
Requires-Python: >=3.9
|
|
11
|
-
Requires-Dist: fastapi>=0.110
|
|
12
|
-
Requires-Dist: psutil>=5.9
|
|
13
|
-
Requires-Dist: pyyaml>=6.0
|
|
14
|
-
Requires-Dist: uvicorn[standard]>=0.27
|
|
15
|
-
Provides-Extra: dev
|
|
16
|
-
Requires-Dist: httpx>=0.27; extra == 'dev'
|
|
17
|
-
Requires-Dist: pytest>=7; extra == 'dev'
|
|
18
|
-
Description-Content-Type: text/markdown
|
|
19
|
-
|
|
20
1
|
# agent-usage-manager
|
|
21
2
|
|
|
22
3
|
A small web dashboard for **headless AI agents** running on a machine —
|
|
23
4
|
OpenClaw, Hermes, Claude Code, Ollama, vLLM, llama.cpp, or anything you name. It
|
|
24
5
|
shows which agents are alive and what they're costing you (CPU, memory, GPU), and
|
|
25
6
|
gives you a **kill button** per agent. Think `htop`, scoped to just your agents —
|
|
26
|
-
the [screenshot below](docs/dashboard.png) is a real run on a fleet node.
|
|
7
|
+
the [screenshot below](https://raw.githubusercontent.com/minglong51/agent-usage-manager/main/docs/dashboard.png) is a real run on a fleet node.
|
|
27
8
|
|
|
28
9
|
No database, no auth framework (one static token file gates the kill switch),
|
|
29
10
|
four direct dependencies (FastAPI, uvicorn, psutil, PyYAML). Runs on macOS and Linux. It is a
|
|
30
11
|
per-node monitor and guarded local control panel: fleet schedulers may consume
|
|
31
12
|
its read-only telemetry, but should own their own scheduling and actuation.
|
|
32
13
|
|
|
33
|
-

|
|
14
|
+

|
|
34
15
|
|
|
35
16
|
*A real run: ten agents grouped by process tree (`+N` = children rolled up),
|
|
36
17
|
per-agent CPU/memory/uptime, launchd-supervised jobs flagged, and a kill button
|
|
@@ -49,6 +30,20 @@ ollama 50122 ● running 3.1 9210 14080 6h 02m ollama ru
|
|
|
49
30
|
> run on Apple Silicon, where per-process GPU stats aren't available so that column
|
|
50
31
|
> is hidden. The UI auto-refreshes every 3s.
|
|
51
32
|
|
|
33
|
+
## Quick start
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
uvx agent-usage-manager # then open http://127.0.0.1:8765 (opens automatically)
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
One command — no install, no virtualenv, no leftovers. Other install options,
|
|
40
|
+
config, and flags: [Install & run](#install--run).
|
|
41
|
+
|
|
42
|
+

|
|
43
|
+
|
|
44
|
+
*Field notes on running agents with discipline go out on the
|
|
45
|
+
[Agent Discipline](https://buttondown.com/minglong51) list — a few emails a month.*
|
|
46
|
+
|
|
52
47
|
## What it does
|
|
53
48
|
|
|
54
49
|
- **One row per agent.** Agents are grouped by process tree — the spawned children
|
|
@@ -64,7 +59,9 @@ ollama 50122 ● running 3.1 9210 14080 6h 02m ollama ru
|
|
|
64
59
|
- **Trends, not just snapshots.** Each row has a CPU sparkline (last ~20 min, sampled
|
|
65
60
|
in the background even with no browser open), plus a **`hot 5m+`** badge when an
|
|
66
61
|
agent has been pegged ≥90% CPU for 5+ minutes, an **`idle 10m+`** badge when a
|
|
67
|
-
long-running agent has done nothing for 10+ minutes
|
|
62
|
+
long-running agent has done nothing for 10+ minutes (suppressed for labels in
|
|
63
|
+
`idle_ok:` — for a fleet that waits for work, idle is the normal state and
|
|
64
|
+
badging it is wallpaper), and a **`churn ×N`** badge when
|
|
68
65
|
the same agent has died young 3+ times in 10 minutes — the states worth investigating
|
|
69
66
|
(runaway, possibly wedged, crash-looping under a supervisor). Churn is what hot/idle
|
|
70
67
|
can't see: a crash-looping process is a fresh pid every poll, so no per-process
|
|
@@ -73,9 +70,15 @@ ollama 50122 ● running 3.1 9210 14080 6h 02m ollama ru
|
|
|
73
70
|
- **Alerts.** A dashboard only helps while you're looking at it. Add an `alerts:`
|
|
74
71
|
block to `agents.yaml` and any badge appearing runs your command (desktop
|
|
75
72
|
notification, Telegram bot, pager — anything) with the details in `$AUM_*` env
|
|
76
|
-
vars.
|
|
73
|
+
vars. `$AUM_MSG` leads with the plain-English verdict ("Codex is crash-looping —
|
|
74
|
+
4 restarts in 10 min"); the machine snapshot trails in brackets. Fires once per
|
|
75
|
+
transition with a cooldown, never from the `list` CLI, and the cooldown is only
|
|
76
|
+
charged when the command exits 0 (a broken notifier doesn't suppress the retry).
|
|
77
77
|
By default only `hot`/`churn`/`leak` alert — `idle` is the normal state of an
|
|
78
78
|
agent fleet that waits for work, so it's opt-in.
|
|
79
|
+
Run `agent-usage-manager test-alert` once after wiring it up: it fires the
|
|
80
|
+
configured command synchronously with a test message and reports the exit
|
|
81
|
+
status, so a broken channel surfaces today instead of during the next incident.
|
|
79
82
|
|
|
80
83
|
```yaml
|
|
81
84
|
alerts:
|
|
@@ -98,11 +101,25 @@ ollama 50122 ● running 3.1 9210 14080 6h 02m ollama ru
|
|
|
98
101
|
- **Config hot-reload.** Edits to `agents.yaml` apply on the next poll, no restart.
|
|
99
102
|
A broken edit keeps the last good config and shows the parse error in the header.
|
|
100
103
|
- **`list` subcommand.** `agent-usage-manager list` (or `list --json`) prints a one-shot
|
|
101
|
-
table to stdout — no server, good for scripts and cron checks.
|
|
104
|
+
table to stdout — no server, good for scripts and cron checks. One-shot mode has no
|
|
105
|
+
history, so the sustained-state flags (`hot`/`idle`/`churn`/`leak`) can never populate
|
|
106
|
+
there (`--json` says `"flags_available": false`); query the running server's
|
|
107
|
+
`/api/agents` when you need flags.
|
|
102
108
|
- **Kill-safe table.** Rows keep a stable order (sorted by label) and never reorder
|
|
103
109
|
while your pointer is over the table, so the kill button can't shift under your
|
|
104
110
|
cursor mid-click.
|
|
105
111
|
|
|
112
|
+
## Why not just htop / Grafana?
|
|
113
|
+
|
|
114
|
+
- **htop sees processes, not agents.** No rollup of a spawned tree under the
|
|
115
|
+
agent that owns it, no agent states (crash-looping, idle, leaking), and an
|
|
116
|
+
unguarded `F9`.
|
|
117
|
+
- **Grafana + Prometheus + an exporter is a stack you operate.** This is one
|
|
118
|
+
command with no database — and it still serves `/metrics` if you want both.
|
|
119
|
+
- **The kill switch is the point.** Allowlist-matched, token-gated, guarded
|
|
120
|
+
against CSRF and DNS rebinding, with an append-only action log — a web page
|
|
121
|
+
that can stop processes needs exactly those.
|
|
122
|
+
|
|
106
123
|
## Product boundary
|
|
107
124
|
|
|
108
125
|
`agent-usage-manager` is intentionally **not** a fleet scheduler, dispatcher, or
|
|
@@ -158,6 +175,25 @@ This is the important part — a web page that can kill processes needs guardrai
|
|
|
158
175
|
also pass `--unsafe-expose`. Exposing the port means one static token is all
|
|
159
176
|
that stands between the network and your agents — put real auth in front
|
|
160
177
|
(reverse proxy + basic auth, SSH tunnel, etc.) before using that flag.
|
|
178
|
+
- **A proxy voids the loopback guarantee.** The production deployment (Ming's
|
|
179
|
+
suite) fronts this loopback bind with `tailscale serve` (`:8448 → 127.0.0.1:8765`),
|
|
180
|
+
so the tailnet reaches the UI despite the loopback bind: reads like
|
|
181
|
+
`/api/agents` are open there, and the static token is the *only* boundary on
|
|
182
|
+
actions. That is deliberate (owner-only tailnet + token-gated actions), but do
|
|
183
|
+
not read "loopback-only" as "not network-reachable" — a proxy in front is not
|
|
184
|
+
a trust boundary.
|
|
185
|
+
- **Fronting it with a proxy needs `AUM_TRUSTED_HOSTS`.** The DNS-rebinding guard
|
|
186
|
+
refuses any request whose `Host` is a non-local DNS name — which is exactly what
|
|
187
|
+
a proxy forwards. Opt that one name in explicitly:
|
|
188
|
+
|
|
189
|
+
```bash
|
|
190
|
+
AUM_TRUSTED_HOSTS=box.your-tailnet.ts.net agent-usage-manager
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
Comma-separated for several. Only add a name nobody else can mint a cert for on
|
|
194
|
+
this node; a name you don't control reopens the hole the guard exists to close.
|
|
195
|
+
Empty by default — out of the box only `localhost`, `127.0.0.1` and `::1` are
|
|
196
|
+
accepted.
|
|
161
197
|
|
|
162
198
|
## Limits & known issues
|
|
163
199
|
|
|
@@ -189,7 +225,7 @@ This is the important part — a web page that can kill processes needs guardrai
|
|
|
189
225
|
- **Windows is untested.** Kill maps to `TerminateProcess` via psutil and may
|
|
190
226
|
work, but CI covers Linux + macOS only.
|
|
191
227
|
|
|
192
|
-
##
|
|
228
|
+
## Install & run
|
|
193
229
|
|
|
194
230
|
**Recommended — one command, nothing to install first:**
|
|
195
231
|
|
|
@@ -281,6 +317,19 @@ and `/metrics` series all use the derived label, so each instance gets its own
|
|
|
281
317
|
state. Sessions that don't match the regex keep their `agents:` label, and the
|
|
282
318
|
key is ignored where tmux isn't installed or running.
|
|
283
319
|
|
|
320
|
+
**Supervised fleets (`launchd_labels:`, macOS)** — the same problem for agents
|
|
321
|
+
that run as launchd jobs and never touch tmux (five `hermes` LaunchAgents all
|
|
322
|
+
landing as "hermes"). The launchd job label is their durable identity:
|
|
323
|
+
|
|
324
|
+
```yaml
|
|
325
|
+
launchd_labels: "^ai\\.hermes\\.(?:gateway-)?(.+)$" # ai.hermes.gateway-frontdoor → frontdoor
|
|
326
|
+
```
|
|
327
|
+
|
|
328
|
+
When a matched root's own launchd job label matches, the first capture group
|
|
329
|
+
(the whole label if there's no group) becomes the row label — with the same
|
|
330
|
+
per-instance churn/alert/metrics identity as `tmux_labels:`. `tmux_labels`
|
|
331
|
+
wins when both apply; roots with no matching job keep their `agents:` label.
|
|
332
|
+
|
|
284
333
|
**Which `agents.yaml` is used** — resolved once at startup, first hit wins:
|
|
285
334
|
|
|
286
335
|
1. `AGENTS_CONFIG=/path/to/agents.yaml` env var (the `--config` flag sets this)
|
|
@@ -349,7 +398,7 @@ Linux (systemd), `~/.config/systemd/user/agent-usage-manager.service`:
|
|
|
349
398
|
[Unit]
|
|
350
399
|
Description=agent usage manager
|
|
351
400
|
[Service]
|
|
352
|
-
ExecStart=%h/agent-usage-manager/.venv/bin/uvicorn app:app --port 8765
|
|
401
|
+
ExecStart=%h/agent-usage-manager/.venv/bin/uvicorn agent_usage_manager.app:app --port 8765
|
|
353
402
|
WorkingDirectory=%h/agent-usage-manager
|
|
354
403
|
Restart=on-failure
|
|
355
404
|
[Install]
|
|
@@ -374,6 +423,35 @@ SIGTERM/SIGKILL on POSIX and TerminateProcess on Windows.
|
|
|
374
423
|
|
|
375
424
|
## Release notes
|
|
376
425
|
|
|
426
|
+
### 0.2.5 — per-instance labels for supervised fleets + the honesty batch
|
|
427
|
+
|
|
428
|
+
- `launchd_labels:` config — per-instance row labels from launchd job labels,
|
|
429
|
+
mirroring `tmux_labels:`. A supervised fleet (several LaunchAgents on one
|
|
430
|
+
binary) no longer collapses into one blurred label for churn tracking, alert
|
|
431
|
+
transitions, and `/metrics`. `idle_ok:` suppression matches the base matcher
|
|
432
|
+
label as well, so a renamed instance (`hermes` → `frontdoor`) keeps its
|
|
433
|
+
class-level suppression.
|
|
434
|
+
- New `test-alert` subcommand: fires the configured `alerts.command` once,
|
|
435
|
+
synchronously, and reports the exit status — proves the alert channel before
|
|
436
|
+
an incident depends on it.
|
|
437
|
+
- `ignore:` now covers ChatGPT.app's embedded Codex helpers (renderer/service
|
|
438
|
+
processes, `Resources/codex` app-server) and the `Codex Computer Use`
|
|
439
|
+
desktop-automation app — they matched the `codex` pattern via the bundle
|
|
440
|
+
name and cluttered the dashboard with GUI plumbing.
|
|
441
|
+
- Config deletion is no longer silent: a vanished `agents.yaml` surfaces as a
|
|
442
|
+
`config_error` ("running on the last good config") instead of looking like
|
|
443
|
+
hot-reload still works.
|
|
444
|
+
- The kill confirm now names the agent (label + command line, not just PID)
|
|
445
|
+
and discloses that plain kill escalates SIGTERM → SIGKILL after 3s.
|
|
446
|
+
- The first-kill token prompt names the server's host (and the ssh one-liner)
|
|
447
|
+
for browsers viewing the dashboard over a tunnel.
|
|
448
|
+
- The DNS-rebinding 403 now names the remedy (use a loopback name or bare IP).
|
|
449
|
+
- `list` says so when flag fields can't populate (one-shot mode has no
|
|
450
|
+
history): a stderr note, and `"flags_available": false` in `--json`.
|
|
451
|
+
- `--host ::1` now produces a valid bracketed IPv6 URL for the browser open.
|
|
452
|
+
- Fixed the README's systemd unit (`uvicorn app:app` could never resolve the
|
|
453
|
+
module; it's `agent_usage_manager.app:app`).
|
|
454
|
+
|
|
377
455
|
### 0.2.4 — stop reporting non-agents
|
|
378
456
|
|
|
379
457
|
- `ignore:` now covers Sparkle's `Autoupdate`/`Updater` (Codex.app's equivalent of
|
|
@@ -436,6 +514,12 @@ SIGTERM/SIGKILL on POSIX and TerminateProcess on Windows.
|
|
|
436
514
|
closed; add `--unsafe-expose` only with auth in front (see
|
|
437
515
|
[Safety](#safety)).
|
|
438
516
|
|
|
517
|
+
## Stay in the loop
|
|
518
|
+
|
|
519
|
+
New tools and field notes on running AI agents with discipline go to the
|
|
520
|
+
[Agent Discipline](https://buttondown.com/minglong51) list first — launch
|
|
521
|
+
notes, operational patterns, early access. A few emails a month at most.
|
|
522
|
+
|
|
439
523
|
## License
|
|
440
524
|
|
|
441
525
|
MIT
|
|
@@ -1,3 +1,8 @@
|
|
|
1
|
+
# Default configuration shipped with the package. It is used only when you have
|
|
2
|
+
# no agents.yaml of your own: `--config` wins, then ./agents.yaml in the current
|
|
3
|
+
# directory, then this file. Copy it somewhere and edit rather than editing it
|
|
4
|
+
# in place — a package upgrade replaces this file.
|
|
5
|
+
#
|
|
1
6
|
# Which processes count as "agents". A process matches if the pattern hits its
|
|
2
7
|
# executable name + first few arguments (case-insensitive substring or regex) —
|
|
3
8
|
# NOT the entire command line, so a process that merely mentions an agent name
|
|
@@ -61,17 +66,42 @@ ignore:
|
|
|
61
66
|
- updater # Sparkle's Updater.app; argv[1] is the app path, so it matches
|
|
62
67
|
- for chrome # "Codex for Chrome" extension host, not the agent
|
|
63
68
|
- tmux attach # tmux clients: `attach -t <session>` carries the agent's name
|
|
69
|
+
- chatgpt codex # ChatGPT.app's embedded Codex helpers ("Codex (Renderer)",
|
|
70
|
+
# "Codex (Service)", Resources/codex app-server) — GUI-app
|
|
71
|
+
# plumbing, matched via the "ChatGPT" bundle prefix; not agents
|
|
72
|
+
- codex computer use # ~/.codex/computer-use desktop-automation helper app
|
|
73
|
+
|
|
74
|
+
# Optional: per-instance labels from launchd job labels — the supervised-fleet
|
|
75
|
+
# counterpart of tmux_labels below. Several LaunchAgents running the same
|
|
76
|
+
# binary all hit one agents: entry and land as indistinguishable rows, and tmux
|
|
77
|
+
# never sees them. When a matched root's own launchd job label matches this
|
|
78
|
+
# regex, the first capture group (the whole label if no group) becomes the row
|
|
79
|
+
# label — churn tracking, alert transitions, and /metrics all get per-instance
|
|
80
|
+
# identity. tmux_labels wins when both apply. Empty/absent = off.
|
|
81
|
+
#
|
|
82
|
+
# launchd_labels: "^com\\.example\\.worker-(.+)$" # → worker-1, worker-2
|
|
64
83
|
|
|
65
84
|
# Optional: per-instance labels from tmux session names. A fleet of identical
|
|
66
|
-
# agents (e.g. several claude-code bots, one per tmux session
|
|
67
|
-
#
|
|
68
|
-
#
|
|
69
|
-
#
|
|
70
|
-
#
|
|
71
|
-
#
|
|
72
|
-
#
|
|
73
|
-
#
|
|
74
|
-
|
|
85
|
+
# agents (e.g. several claude-code bots, one per tmux session) all hit one
|
|
86
|
+
# agents: entry above and land as N indistinguishable rows — their cmdlines
|
|
87
|
+
# can't tell them apart, because matching deliberately sees only the executable
|
|
88
|
+
# + first args. The tmux session each one runs in IS its identity: when a
|
|
89
|
+
# matched root (or an ancestor) is a tmux pane whose session name matches this
|
|
90
|
+
# regex, the row is labeled with the first capture group (the whole session
|
|
91
|
+
# name if there is no group). Sessions that don't match keep their agents:
|
|
92
|
+
# label, so incidental tmux use never renames rows.
|
|
93
|
+
#
|
|
94
|
+
# tmux_labels: "^bot-(.+)$" # session bot-worker1 → row label worker1
|
|
95
|
+
|
|
96
|
+
# Labels whose idle state is NORMAL — agents that wait for work (bots parked on
|
|
97
|
+
# a chat poll, gateways waiting for requests). They get no "idle" badge: badging
|
|
98
|
+
# the whole waiting fleet wallpapers the dashboard and trains badge-blindness.
|
|
99
|
+
# Same reasoning as `idle` being opt-in for alerts below. Case-insensitive
|
|
100
|
+
# substrings of the row label (incl. tmux-derived ones), like ignore:.
|
|
101
|
+
#
|
|
102
|
+
# idle_ok:
|
|
103
|
+
# - gateway
|
|
104
|
+
# - worker
|
|
75
105
|
|
|
76
106
|
# GPU sampling: nvidia-smi is used automatically when present (Linux/NVIDIA).
|
|
77
107
|
# On Apple Silicon there is no per-process GPU API, so the GPU column is blank.
|
|
@@ -85,24 +115,15 @@ tmux_labels: "^bot-(.+)$" # session bot-coder_1 → row label coder_1
|
|
|
85
115
|
# fleet of agents that wait for work, idle is the NORMAL state, and alerting on
|
|
86
116
|
# it floods the channel every time the server restarts and re-learns the fleet.
|
|
87
117
|
#
|
|
88
|
-
#
|
|
118
|
+
# Nothing is wired by default — uncomment and point it at your own channel, then
|
|
119
|
+
# prove it with `agent-usage-manager test-alert` before a real badge depends on
|
|
120
|
+
# it. Two things worth knowing before you turn `hot` on: for inference agents
|
|
121
|
+
# pegged CPU IS the job, and spawn-heavy agents look like churn while working.
|
|
122
|
+
#
|
|
89
123
|
# alerts:
|
|
90
124
|
# command: 'terminal-notifier -title agent-usage-manager -message "$AUM_MSG"'
|
|
91
125
|
# cooldown: 600 # seconds, default 600
|
|
92
126
|
# flags: [hot, churn, leak] # default; add idle only if you really want it
|
|
93
|
-
#
|
|
94
|
-
#
|
|
95
|
-
#
|
|
96
|
-
# positives — "codex churn" during legitimate spawn work and "claude-code hot"
|
|
97
|
-
# during a working session; for inference agents, pegged-CPU IS the job.
|
|
98
|
-
# Badges stay live on the dashboard + machinery panel; the feed line rides the
|
|
99
|
-
# next digest. Genuine breakage has louder, independent signals (mcp-autoheal
|
|
100
|
-
# GAVE UP, cron-health, sync-bot liveness). `hot` dropped from alerting
|
|
101
|
-
# entirely; notify.py is a `uv run` script, hence the inline PATH prefix.
|
|
102
|
-
alerts:
|
|
103
|
-
command: 'PATH="$HOME/.local/bin:$PATH"; "$HOME/.claude/skills/_finance_lib/scripts/notify.py" --plain --priority digest --source usage-manager --feed-only --text "$AUM_MSG"'
|
|
104
|
-
cooldown: 7200 # 2 h per (agent, flag)
|
|
105
|
-
flags: [churn, leak]
|
|
106
|
-
leak_floor_mb: 1536 # bots legitimately ratchet RSS as session
|
|
107
|
-
# context grows; only push leak alerts once
|
|
108
|
-
# the absolute footprint is actually large
|
|
127
|
+
# leak_floor_mb: 1536 # only alert on leak once the absolute footprint
|
|
128
|
+
# # is large — long-running agents legitimately
|
|
129
|
+
# # ratchet RSS as their context grows
|