agentmon 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +14 -0
- data/LICENSE.txt +21 -0
- data/README.md +105 -0
- data/exe/agentmon +18 -0
- data/lib/agentmon/commands/cli_options.rb +21 -0
- data/lib/agentmon/commands/footprint.rb +81 -0
- data/lib/agentmon/commands/memory.rb +110 -0
- data/lib/agentmon/commands/record.rb +118 -0
- data/lib/agentmon/commands/report.rb +127 -0
- data/lib/agentmon/commands/sessions.rb +98 -0
- data/lib/agentmon/commands/top.rb +87 -0
- data/lib/agentmon/commands/tree.rb +96 -0
- data/lib/agentmon/darwin.rb +235 -0
- data/lib/agentmon/engine.rb +115 -0
- data/lib/agentmon/focus.rb +88 -0
- data/lib/agentmon/metrics/memory.rb +76 -0
- data/lib/agentmon/metrics/network.rb +245 -0
- data/lib/agentmon/metrics/pressure_drivers.rb +57 -0
- data/lib/agentmon/metrics/process_rates.rb +65 -0
- data/lib/agentmon/metrics/process_rows.rb +41 -0
- data/lib/agentmon/metrics/session_ledger.rb +160 -0
- data/lib/agentmon/metrics/session_memory.rb +84 -0
- data/lib/agentmon/metrics/session_names.rb +169 -0
- data/lib/agentmon/metrics/sessions.rb +80 -0
- data/lib/agentmon/metrics/system.rb +114 -0
- data/lib/agentmon/model.rb +263 -0
- data/lib/agentmon/probes/cwd.rb +32 -0
- data/lib/agentmon/probes/memory.rb +139 -0
- data/lib/agentmon/probes/network.rb +326 -0
- data/lib/agentmon/probes/processes.rb +72 -0
- data/lib/agentmon/probes/system.rb +134 -0
- data/lib/agentmon/program.rb +55 -0
- data/lib/agentmon/reading.rb +83 -0
- data/lib/agentmon/recorders/memory.rb +19 -0
- data/lib/agentmon/recorders/sessions.rb +40 -0
- data/lib/agentmon/registry.rb +155 -0
- data/lib/agentmon/sampler.rb +50 -0
- data/lib/agentmon/store.rb +86 -0
- data/lib/agentmon/ui/connections.rb +75 -0
- data/lib/agentmon/ui/interaction.rb +94 -0
- data/lib/agentmon/ui/memory.rb +125 -0
- data/lib/agentmon/ui/process_actions.rb +49 -0
- data/lib/agentmon/ui/process_detail.rb +148 -0
- data/lib/agentmon/ui/process_network.rb +13 -0
- data/lib/agentmon/ui/process_scopes.rb +32 -0
- data/lib/agentmon/ui/process_waits.rb +99 -0
- data/lib/agentmon/ui/processes.rb +42 -0
- data/lib/agentmon/ui/session_focus.rb +126 -0
- data/lib/agentmon/ui/session_memory.rb +78 -0
- data/lib/agentmon/ui/sessions.rb +130 -0
- data/lib/agentmon/ui/theme.rb +49 -0
- data/lib/agentmon/ui.rb +80 -0
- data/lib/agentmon/version.rb +5 -0
- data/lib/agentmon/views/dense.rb +157 -0
- data/lib/agentmon/views/history.rb +62 -0
- data/lib/agentmon/views/signals.rb +279 -0
- data/lib/agentmon/views/visual.rb +172 -0
- data/lib/agentmon/views/widgets.rb +352 -0
- data/lib/agentmon/views.rb +875 -0
- data/lib/agentmon.rb +38 -0
- metadata +135 -0
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# reading[:session_ledger]: every agent session alive now or ended in the last KEEP_ENDED seconds,
|
|
4
|
+
# as Session records (lib/agentmon/model.rb) with current values and lifetime totals. Stateful: it
|
|
5
|
+
# sees every sample the Engine takes.
|
|
6
|
+
#
|
|
7
|
+
# How the totals are counted, and what they can't see:
|
|
8
|
+
#
|
|
9
|
+
# - cpu_seconds adds, for each member process, the growth of its own + reaped-children CPU time
|
|
10
|
+
# between samples (a member seen for the first time counts all of it, so a session that started
|
|
11
|
+
# before agentmon still shows its whole life). Child time matters: the kernel adds a process's
|
|
12
|
+
# CPU time to its parent's when the parent reaps it, so short-lived tools (git, rg) that start
|
|
13
|
+
# and exit between two samples are still counted, through their parent.
|
|
14
|
+
# That also means a member that exits is about to be counted again inside its parent's child
|
|
15
|
+
# time, so its last seen total goes into the session's `pending` pool and is subtracted from the
|
|
16
|
+
# session's next growth (for PENDING_SAMPLES samples; then it expires, so a parent outside the
|
|
17
|
+
# tree or a zombie nobody reaps can't swallow real growth). When a CLI session ends under a parent in another session (Claude Code
|
|
18
|
+
# under the desktop app), its root's total goes into that session's pool the same way.
|
|
19
|
+
# A member that leaves the tree alive (an orphan reparented to launchd) keeps what it was
|
|
20
|
+
# counted and stops counting.
|
|
21
|
+
# - bytes_read / bytes_written add each member's growth between samples (first seen: all of it).
|
|
22
|
+
# Disk I/O isn't passed to parents, so a process that starts and exits between two samples
|
|
23
|
+
# isn't seen: these are lower bounds.
|
|
24
|
+
# - peak_footprint is the largest footprint sum (readable members) seen in any sample.
|
|
25
|
+
module Agentmon
|
|
26
|
+
module Metrics
|
|
27
|
+
class SessionLedger
|
|
28
|
+
KEEP_ENDED = 15 * 60
|
|
29
|
+
# Samples an exited member's CPU time waits to reappear as its parent's child time.
|
|
30
|
+
PENDING_SAMPLES = 5
|
|
31
|
+
|
|
32
|
+
# Mutable per-session accumulator (the metric's state); `to_session` makes the record.
|
|
33
|
+
Account = Struct.new(:info, :first_seen_at, :last_seen_at, :ended_at, :pending, :cpu_seconds, :bytes_read,
|
|
34
|
+
:bytes_written, :peak_footprint, :now, :parent_session) do
|
|
35
|
+
def alive? = ended_at.nil?
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
def initialize(state) = @state = state
|
|
39
|
+
|
|
40
|
+
def call(reading)
|
|
41
|
+
accounts = (@state[:accounts] ||= {})
|
|
42
|
+
seen = @state[:seen] || {} # identity => [session id, cpu total, disk read, disk written]
|
|
43
|
+
map = reading[:sessions]
|
|
44
|
+
# Attribution failed this sample: keep every account as it was rather than end them all.
|
|
45
|
+
return sessions(accounts) unless map
|
|
46
|
+
|
|
47
|
+
rates = reading[:process_rates] || {}
|
|
48
|
+
processes = reading.sample[:processes] || []
|
|
49
|
+
live = processes.select(&:readable).to_h { |p| [p.identity, true] }
|
|
50
|
+
at = reading.at
|
|
51
|
+
|
|
52
|
+
end_missing(accounts, map, at)
|
|
53
|
+
hand_over_exits(accounts, seen, live)
|
|
54
|
+
members = processes.group_by { |p| map.by_pid[p.pid] }
|
|
55
|
+
@state[:seen] = next_seen = {}
|
|
56
|
+
map.sessions.each do |info|
|
|
57
|
+
account = accounts[info.id] ||= Account.new(info:, first_seen_at: at, pending: [], cpu_seconds: 0.0,
|
|
58
|
+
bytes_read: 0, bytes_written: 0, peak_footprint: 0)
|
|
59
|
+
update(account, info, members.fetch(info.id, []), rates, seen, next_seen, at, map)
|
|
60
|
+
end
|
|
61
|
+
accounts.delete_if { |_, a| a.ended_at && at - a.ended_at > KEEP_ENDED }
|
|
62
|
+
sessions(accounts)
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
private
|
|
66
|
+
|
|
67
|
+
def sessions(accounts)
|
|
68
|
+
accounts.values.sort_by { |a| [a.info.started_at || a.first_seen_at, a.info.id] }.map { |a| to_session(a) }
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
# Pending CPU seconds wait PENDING_SAMPLES samples for the parent to report them as child time
|
|
72
|
+
# (it reaps within milliseconds); if it never does (reaped outside the session, a zombie
|
|
73
|
+
# nobody reaps), they expire so the session's real growth isn't swallowed.
|
|
74
|
+
def expect_handover(account, seconds) = account.pending << [seconds, PENDING_SAMPLES]
|
|
75
|
+
|
|
76
|
+
# Takes up to `grown` seconds out of the pending pool, oldest first; returns what it took.
|
|
77
|
+
def absorb(account, grown)
|
|
78
|
+
taken = 0.0
|
|
79
|
+
account.pending.each do |entry|
|
|
80
|
+
take = [entry[0], grown - taken].min
|
|
81
|
+
entry[0] -= take
|
|
82
|
+
taken += take
|
|
83
|
+
end
|
|
84
|
+
account.pending.each { |entry| entry[1] -= 1 }
|
|
85
|
+
account.pending.reject! { |amount, left| amount <= 0 || left <= 0 }
|
|
86
|
+
taken
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
def end_missing(accounts, map, _at)
|
|
90
|
+
alive = map.sessions.to_h { |s| [s.id, true] }
|
|
91
|
+
accounts.each_value do |account|
|
|
92
|
+
next if alive[account.info.id] || account.ended_at
|
|
93
|
+
|
|
94
|
+
account.ended_at = account.last_seen_at
|
|
95
|
+
account.now = nil
|
|
96
|
+
end
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
# Members that exited since the last sample will show up again as their parent's child time.
|
|
100
|
+
def hand_over_exits(accounts, seen, live)
|
|
101
|
+
seen.each do |identity, (session_id, total)|
|
|
102
|
+
next if live[identity]
|
|
103
|
+
|
|
104
|
+
account = accounts[session_id] or next
|
|
105
|
+
if account.alive?
|
|
106
|
+
expect_handover(account, total)
|
|
107
|
+
elsif identity[0] == account.info.root_pid && account.info.kind == :cli
|
|
108
|
+
heir = accounts[account.parent_session]
|
|
109
|
+
expect_handover(heir, total) if heir&.alive?
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
def update(account, info, procs, rates, seen, next_seen, at, map)
|
|
115
|
+
grown = 0.0
|
|
116
|
+
readable = procs.select(&:readable)
|
|
117
|
+
readable.each do |p|
|
|
118
|
+
total = p.cpu_time + p.child_cpu_time
|
|
119
|
+
was = seen[p.identity]
|
|
120
|
+
was = nil unless was && was[0] == info.id
|
|
121
|
+
grown += was ? [total - was[1], 0.0].max : total
|
|
122
|
+
account.bytes_read += was ? [p.disk_read - was[2], 0].max : p.disk_read
|
|
123
|
+
account.bytes_written += was ? [p.disk_written - was[3], 0].max : p.disk_written
|
|
124
|
+
next_seen[p.identity] = [info.id, total, p.disk_read, p.disk_written]
|
|
125
|
+
end
|
|
126
|
+
account.cpu_seconds += grown - absorb(account, grown)
|
|
127
|
+
|
|
128
|
+
footprint = readable.sum(&:footprint)
|
|
129
|
+
account.peak_footprint = [account.peak_footprint, footprint].max
|
|
130
|
+
account.info = info
|
|
131
|
+
account.last_seen_at = at
|
|
132
|
+
account.ended_at = nil
|
|
133
|
+
account.parent_session = map.by_pid[procs.find { |p| p.pid == info.root_pid }&.ppid]
|
|
134
|
+
account.now = {
|
|
135
|
+
processes: procs.size,
|
|
136
|
+
cpu: procs.sum { |p| rates[p.pid]&.cpu || 0.0 },
|
|
137
|
+
footprint:,
|
|
138
|
+
resident: procs.sum { |p| p.resident || 0 },
|
|
139
|
+
read_rate: procs.sum { |p| rates[p.pid]&.read_rate || 0.0 },
|
|
140
|
+
write_rate: procs.sum { |p| rates[p.pid]&.write_rate || 0.0 }
|
|
141
|
+
}
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def to_session(account)
|
|
145
|
+
info = account.info
|
|
146
|
+
now = account.now || { processes: 0, cpu: 0.0, footprint: 0, resident: 0, read_rate: 0.0, write_rate: 0.0 }
|
|
147
|
+
Session.new(
|
|
148
|
+
id: info.id, kind: info.kind, name: info.name, root_pid: info.root_pid, label: info.label, cwd: info.cwd,
|
|
149
|
+
started_at: info.started_at || account.first_seen_at, first_seen_at: account.first_seen_at,
|
|
150
|
+
last_seen_at: account.last_seen_at, ended_at: account.ended_at,
|
|
151
|
+
peak_footprint: account.peak_footprint, cpu_seconds: account.cpu_seconds,
|
|
152
|
+
bytes_read: account.bytes_read, bytes_written: account.bytes_written,
|
|
153
|
+
title: info.title, status: info.status, threads: info.threads, **now
|
|
154
|
+
)
|
|
155
|
+
end
|
|
156
|
+
end
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
metric(:session_ledger) { |reading, state| Metrics::SessionLedger.new(state).call(reading) }
|
|
160
|
+
end
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# reading[:session_memory]: how much memory each alive agent session holds, as SessionMemory
|
|
4
|
+
# records (lib/agentmon/model.rb), largest footprint first. For example:
|
|
5
|
+
#
|
|
6
|
+
# claude 4242 · repo footprint 2.1G resident 2.4G wired 1.0M 26.3% +1.5M/s 10 pageins/s
|
|
7
|
+
# Claude 59334 footprint 900M resident 1.1G wired 0B 11.0% -12K/s 0 pageins/s
|
|
8
|
+
#
|
|
9
|
+
# - footprint, resident, wired, pageins, pagein_rate, fault_rate: sums over the session's live
|
|
10
|
+
# members in reading[:process_rows], each pid once. A member whose value is unknown (nil: an
|
|
11
|
+
# unreadable process, or a rate before two samples) is skipped; a sum with no known member is
|
|
12
|
+
# nil, never 0. Members are rows whose session_id is the session's, so a process tree counts
|
|
13
|
+
# each process once (sessions.rb puts every pid in at most one session).
|
|
14
|
+
# - processes: the live members seen in process_rows.
|
|
15
|
+
# - peak_footprint: the ledger's (Session#peak_footprint), the largest footprint sum over its life.
|
|
16
|
+
# - share: footprint as a percent of used memory (reading[:memory].used); nil while memory is
|
|
17
|
+
# unknown.
|
|
18
|
+
# - growth_rate: footprint change in bytes per second over the last WINDOW seconds of samples, on
|
|
19
|
+
# the monotonic clock; 0.0 until a session has two samples (as pressure_drivers does), nil while
|
|
20
|
+
# its footprint is unknown. Only samples with the same readable members (pids with a known
|
|
21
|
+
# footprint) are compared, so a member joining, leaving or turning unreadable restarts the
|
|
22
|
+
# window rather than showing as growth.
|
|
23
|
+
#
|
|
24
|
+
# Compressed and swapped bytes per process need task_for_pid (root), so they aren't here; the
|
|
25
|
+
# machine's are in reading[:memory]. Empty when the ledger is nil; when reading[:process_rows] is
|
|
26
|
+
# nil, one record per alive session with nil sums, since no member is known. Stateful: a short
|
|
27
|
+
# footprint history per session id, dropped when the session is no longer alive.
|
|
28
|
+
module Agentmon
|
|
29
|
+
module Metrics
|
|
30
|
+
class SessionFootprint
|
|
31
|
+
WINDOW = 60.0
|
|
32
|
+
|
|
33
|
+
def initialize(state) = @state = state
|
|
34
|
+
|
|
35
|
+
def call(reading)
|
|
36
|
+
history = (@state[:history] ||= {}) # session id => [[mono, footprint], ...], oldest first
|
|
37
|
+
alive = (reading[:session_ledger] || []).select(&:alive?)
|
|
38
|
+
history.select! { |id, _| alive.any? { |s| s.id == id } }
|
|
39
|
+
members = (reading[:process_rows] || []).uniq(&:pid).group_by(&:session_id)
|
|
40
|
+
used = reading[:memory]&.used
|
|
41
|
+
mono = reading.sample.mono
|
|
42
|
+
|
|
43
|
+
alive.map do |session|
|
|
44
|
+
rows = members[session.id] || []
|
|
45
|
+
footprint = sum(rows, :footprint)
|
|
46
|
+
SessionMemory.new(
|
|
47
|
+
session_id: session.id, label: session.label, processes: rows.size, footprint:,
|
|
48
|
+
resident: sum(rows, :resident), wired: sum(rows, :wired), peak_footprint: session.peak_footprint,
|
|
49
|
+
share: footprint && used&.positive? ? footprint * 100.0 / used : nil,
|
|
50
|
+
growth_rate: growth(history[session.id] ||= [], mono, footprint, readable(rows)),
|
|
51
|
+
pageins: sum(rows, :pageins), pagein_rate: sum(rows, :pagein_rate), fault_rate: sum(rows, :fault_rate)
|
|
52
|
+
)
|
|
53
|
+
end.sort_by { |m| [-(m.footprint || 0), m.session_id] }
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
private
|
|
57
|
+
|
|
58
|
+
# The sum of the known values; nil when none is known.
|
|
59
|
+
def sum(rows, field)
|
|
60
|
+
known = rows.filter_map { |r| r.public_send(field) }
|
|
61
|
+
known.empty? ? nil : known.sum
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# The pids whose footprint is known, sorted.
|
|
65
|
+
def readable(rows) = rows.reject { |r| r.footprint.nil? }.map(&:pid).sort
|
|
66
|
+
|
|
67
|
+
# Adds this sample to a session's points, forgets those older than WINDOW or with other
|
|
68
|
+
# readable members, and returns the rate from the oldest left to this one; nil (and nothing
|
|
69
|
+
# recorded) while the footprint is unknown.
|
|
70
|
+
def growth(points, mono, footprint, pids)
|
|
71
|
+
return nil if footprint.nil?
|
|
72
|
+
|
|
73
|
+
points.pop if points.last && points.last[0] >= mono # the same sample seen again
|
|
74
|
+
points.clear if points.last && points.last[2] != pids # members changed: restart the window
|
|
75
|
+
points << [mono, footprint, pids]
|
|
76
|
+
points.shift while mono - points.first[0] > WINDOW
|
|
77
|
+
since, was, = points.first
|
|
78
|
+
mono > since ? (footprint - was) / (mono - since) : 0.0
|
|
79
|
+
end
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
metric(:session_memory) { |reading, state| Metrics::SessionFootprint.new(state).call(reading) }
|
|
84
|
+
end
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
# reading[:session_names]: { root pid => SessionName } for the agent CLI processes whose human
|
|
6
|
+
# name (or status) is known. Metrics::Sessions puts the name at the front of the session label.
|
|
7
|
+
#
|
|
8
|
+
# Where names come from (no subprocess, ever):
|
|
9
|
+
#
|
|
10
|
+
# - Claude Code writes ~/.claude/sessions/<pid>.json for each CLI session ("name", "status"
|
|
11
|
+
# busy/idle, "startedAt" ms, ...). Re-parsed only when its mtime changes.
|
|
12
|
+
# - One codex process (the ChatGPT app's `codex app-server`, or the codex CLI) hosts threads and
|
|
13
|
+
# keeps each one's rollout file open: $CODEX_HOME/sessions/YYYY/MM/DD/rollout-<time>-<uuid>.jsonl.
|
|
14
|
+
# $CODEX_HOME/session_index.jsonl maps thread id => thread_name (later lines win); it is
|
|
15
|
+
# append-only, so it is read incrementally from the last byte offset (from 0 if it shrank).
|
|
16
|
+
# The open files come from Darwin.open_paths, once per live codex pid. The title is the newest
|
|
17
|
+
# (rollout mtime) named open thread, plus " +N" for its N other open threads.
|
|
18
|
+
#
|
|
19
|
+
# Cost: nothing per sample. It refreshes at most every REFRESH seconds (monotonic); in between
|
|
20
|
+
# it returns its last value. Unreadable, missing or garbled files mean no name, never an error.
|
|
21
|
+
module Agentmon
|
|
22
|
+
module Metrics
|
|
23
|
+
module SessionNames
|
|
24
|
+
REFRESH = 10.0
|
|
25
|
+
CLI = /\A(claude|codex)\z/
|
|
26
|
+
ROLLOUT = /rollout-.*-(\h{8}-\h{4}-\h{4}-\h{4}-\h{12})\.jsonl\z/
|
|
27
|
+
# A Claude session file written before this process started belongs to an earlier process
|
|
28
|
+
# that had the same pid (seconds of slack for clock rounding).
|
|
29
|
+
STALE_SLACK = 60
|
|
30
|
+
|
|
31
|
+
class << self
|
|
32
|
+
# Injectable for tests (test_helper points them at an empty dir and a nil reader).
|
|
33
|
+
attr_writer :claude_dir, :codex_home, :open_paths
|
|
34
|
+
|
|
35
|
+
def claude_dir = @claude_dir || File.join(Dir.home, ".claude", "sessions")
|
|
36
|
+
def codex_home = @codex_home || ENV["CODEX_HOME"] || File.join(Dir.home, ".codex")
|
|
37
|
+
def open_paths = @open_paths || Darwin.method(:open_paths)
|
|
38
|
+
|
|
39
|
+
# The metric: the last value until REFRESH seconds have passed, then a refresh.
|
|
40
|
+
def call(reading, state)
|
|
41
|
+
mono = reading.sample.mono
|
|
42
|
+
return state[:value] if state[:at] && mono - state[:at] < REFRESH
|
|
43
|
+
|
|
44
|
+
state[:at] = mono
|
|
45
|
+
state[:value] = refresh(reading.sample[:processes] || [], state)
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
# { pid => SessionName } for the live claude/codex roots in `processes`.
|
|
49
|
+
def refresh(processes, state)
|
|
50
|
+
by_pid = processes.to_h { |p| [p.pid, p] }
|
|
51
|
+
cli = ->(p) { p && p.name.to_s.match?(CLI) }
|
|
52
|
+
roots = processes.select { |p| cli[p] && !cli[by_pid[p.ppid]] } # under another CLI: not a root
|
|
53
|
+
names = {}
|
|
54
|
+
claude(roots.select { |p| p.name == "claude" }, state, names)
|
|
55
|
+
codex = roots.select { |p| p.name == "codex" }
|
|
56
|
+
codex(codex, state, names) unless codex.empty?
|
|
57
|
+
names
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
private
|
|
61
|
+
|
|
62
|
+
def claude(processes, state, names)
|
|
63
|
+
cache = (state[:claude] ||= {}) # path => [mtime, SessionName or nil, startedAt seconds]
|
|
64
|
+
live = {}
|
|
65
|
+
processes.each do |p|
|
|
66
|
+
path = File.join(claude_dir, "#{p.pid}.json")
|
|
67
|
+
live[path] = true
|
|
68
|
+
entry = claude_file(path, cache)
|
|
69
|
+
next unless entry
|
|
70
|
+
|
|
71
|
+
_, name, started = entry
|
|
72
|
+
next if started && p.started_at && started < p.started_at - STALE_SLACK
|
|
73
|
+
|
|
74
|
+
names[p.pid] = name if name
|
|
75
|
+
end
|
|
76
|
+
cache.select! { |path, _| live[path] }
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def claude_file(path, cache)
|
|
80
|
+
mtime = File.mtime(path)
|
|
81
|
+
return cache[path] if cache[path] && cache[path][0] == mtime
|
|
82
|
+
|
|
83
|
+
cache[path] = [mtime, *parse_claude(File.read(path))]
|
|
84
|
+
rescue SystemCallError, IOError
|
|
85
|
+
cache.delete(path)
|
|
86
|
+
nil
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# [SessionName or nil, startedAt in epoch seconds or nil]
|
|
90
|
+
def parse_claude(text)
|
|
91
|
+
json = JSON.parse(text)
|
|
92
|
+
return [nil, nil] unless json.is_a?(Hash)
|
|
93
|
+
|
|
94
|
+
title = json["name"].is_a?(String) && !json["name"].strip.empty? ? json["name"].strip : nil
|
|
95
|
+
status = json["status"].is_a?(String) ? json["status"] : nil
|
|
96
|
+
started = json["startedAt"].is_a?(Numeric) ? json["startedAt"] / 1000.0 : nil
|
|
97
|
+
[title || status ? SessionName.new(title:, status:, threads: []) : nil, started]
|
|
98
|
+
rescue JSON::ParserError, EncodingError
|
|
99
|
+
[nil, nil]
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
def codex(processes, state, names)
|
|
103
|
+
index = read_index(state)
|
|
104
|
+
reader = open_paths
|
|
105
|
+
processes.each do |p|
|
|
106
|
+
paths = begin
|
|
107
|
+
reader.call(p.pid)
|
|
108
|
+
rescue StandardError
|
|
109
|
+
nil
|
|
110
|
+
end
|
|
111
|
+
name = codex_name(paths || [], index)
|
|
112
|
+
names[p.pid] = name if name
|
|
113
|
+
end
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# The open rollout files' threads, newest first, named from the index.
|
|
117
|
+
def codex_name(paths, index)
|
|
118
|
+
threads = paths.filter_map { |path| (m = ROLLOUT.match(path)) && [m[1], path] }.uniq(&:first)
|
|
119
|
+
dated = threads.filter_map do |id, path|
|
|
120
|
+
[id, File.mtime(path)]
|
|
121
|
+
rescue SystemCallError
|
|
122
|
+
nil
|
|
123
|
+
end
|
|
124
|
+
return nil if dated.empty?
|
|
125
|
+
|
|
126
|
+
named = dated.sort_by { |_, mtime| -mtime.to_f }.filter_map { |id, _| index[id] }
|
|
127
|
+
return nil if named.empty?
|
|
128
|
+
|
|
129
|
+
more = dated.size - 1
|
|
130
|
+
SessionName.new(title: more.positive? ? "#{named.first} +#{more}" : named.first, status: nil, threads: named)
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
# The thread id => name map, after reading what was appended since the last refresh.
|
|
134
|
+
def read_index(state)
|
|
135
|
+
index = (state[:index] ||= { offset: 0, names: {} })
|
|
136
|
+
path = File.join(codex_home, "session_index.jsonl")
|
|
137
|
+
size = File.size(path)
|
|
138
|
+
index.merge!(offset: 0, names: {}) if size < index[:offset]
|
|
139
|
+
return index[:names] if size == index[:offset]
|
|
140
|
+
|
|
141
|
+
chunk = File.open(path, "rb") do |f|
|
|
142
|
+
f.seek(index[:offset])
|
|
143
|
+
f.read(size - index[:offset])
|
|
144
|
+
end.to_s
|
|
145
|
+
complete = chunk.rindex("\n")
|
|
146
|
+
return index[:names] unless complete # a line still being written: next time
|
|
147
|
+
|
|
148
|
+
chunk.byteslice(0, complete + 1).each_line { |line| index_line(line, index[:names]) }
|
|
149
|
+
index[:offset] += complete + 1
|
|
150
|
+
index[:names]
|
|
151
|
+
rescue SystemCallError, IOError
|
|
152
|
+
index[:names]
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
def index_line(line, names)
|
|
156
|
+
json = JSON.parse(line.force_encoding(Encoding::UTF_8))
|
|
157
|
+
return unless json.is_a?(Hash) && json["id"].is_a?(String)
|
|
158
|
+
|
|
159
|
+
name = json["thread_name"]
|
|
160
|
+
names[json["id"]] = name.strip if name.is_a?(String) && !name.strip.empty?
|
|
161
|
+
rescue JSON::ParserError, EncodingError
|
|
162
|
+
nil
|
|
163
|
+
end
|
|
164
|
+
end
|
|
165
|
+
end
|
|
166
|
+
end
|
|
167
|
+
|
|
168
|
+
metric(:session_names) { |reading, state| Metrics::SessionNames.call(reading, state) }
|
|
169
|
+
end
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# reading[:sessions]: a SessionMap of the agent sessions alive in this sample and which session
|
|
4
|
+
# each pid belongs to.
|
|
5
|
+
#
|
|
6
|
+
# A process belongs to its outermost `claude`/`codex` CLI ancestor (itself included), so shells,
|
|
7
|
+
# tools, MCP servers and sub-agents count toward the session that started them. With no CLI
|
|
8
|
+
# ancestor, it belongs to the topmost agent desktop app ancestor (Claude, ChatGPT/Codex), so all
|
|
9
|
+
# of an app's helpers are one session. Claude Code sessions that the desktop app launches are CLI
|
|
10
|
+
# sessions of their own, not part of the app's. Anything else is in no session.
|
|
11
|
+
module Agentmon
|
|
12
|
+
module Metrics
|
|
13
|
+
module Sessions
|
|
14
|
+
CLI = /\A(claude|codex)\z/
|
|
15
|
+
APP = /claude|codex|chatgpt/i
|
|
16
|
+
|
|
17
|
+
module_function
|
|
18
|
+
|
|
19
|
+
# `names`: reading[:session_names], { root pid => SessionName } (nil or {} when unknown).
|
|
20
|
+
def call(processes, cwds, names = {})
|
|
21
|
+
processes ||= []
|
|
22
|
+
cwds ||= {}
|
|
23
|
+
names ||= {}
|
|
24
|
+
by_pid = processes.to_h { |p| [p.pid, p] }
|
|
25
|
+
chains = {}
|
|
26
|
+
sessions = {}
|
|
27
|
+
map = {}
|
|
28
|
+
processes.each do |p|
|
|
29
|
+
cli, app = chain(p, by_pid, chains)
|
|
30
|
+
root = cli || app
|
|
31
|
+
next unless root
|
|
32
|
+
|
|
33
|
+
info = sessions[root.pid] ||= info(root, cli ? :cli : :app, cwds, names[root.pid])
|
|
34
|
+
map[p.pid] = info.id
|
|
35
|
+
end
|
|
36
|
+
SessionMap.new(sessions: sessions.values, by_pid: map)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# [outermost CLI ancestor-or-self, topmost app ancestor-or-self], memoized per pid. Walks up
|
|
40
|
+
# iteratively and stops at a cycle (pid 0 is its own parent).
|
|
41
|
+
def chain(process, by_pid, memo)
|
|
42
|
+
path = []
|
|
43
|
+
seen = {}
|
|
44
|
+
node = process
|
|
45
|
+
while node && !memo.key?(node.pid) && !seen[node.pid]
|
|
46
|
+
seen[node.pid] = true
|
|
47
|
+
path << node
|
|
48
|
+
node = by_pid[node.ppid]
|
|
49
|
+
end
|
|
50
|
+
cli, app = node && memo[node.pid]
|
|
51
|
+
path.reverse_each do |n|
|
|
52
|
+
cli ||= n if n.name.match?(CLI)
|
|
53
|
+
app ||= n if n.name.match?(APP)
|
|
54
|
+
memo[n.pid] = [cli, app]
|
|
55
|
+
end
|
|
56
|
+
memo[process.pid]
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# Label: "claude 4242 · repo" (directory of a CLI root), "Claude 300" (app); with a human
|
|
60
|
+
# name first: "Agentmon r2ui integration · claude 4242 · repo".
|
|
61
|
+
def info(root, kind, cwds, name = nil)
|
|
62
|
+
cwd = cwds[root.pid]
|
|
63
|
+
id = [root.name, root.pid, root.started_at&.to_i].compact.join("-")
|
|
64
|
+
label = if kind == :cli && cwd && cwd != "/"
|
|
65
|
+
"#{root.name} #{root.pid} · #{File.basename(cwd)}"
|
|
66
|
+
else
|
|
67
|
+
"#{root.name} #{root.pid}"
|
|
68
|
+
end
|
|
69
|
+
title = name&.title
|
|
70
|
+
label = "#{title} · #{label}" if title && !title.empty?
|
|
71
|
+
SessionInfo.new(id:, kind:, name: root.name, root_pid: root.pid, label:, cwd:, started_at: root.started_at,
|
|
72
|
+
title:, status: name&.status, threads: name&.threads || [])
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
metric(:sessions) do |reading|
|
|
78
|
+
Metrics::Sessions.call(reading.sample[:processes], reading.sample[:cwd], reading[:session_names])
|
|
79
|
+
end
|
|
80
|
+
end
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Agentmon
|
|
4
|
+
# reading[:system]: the whole machine's CPU, load, network and disk now, as a SystemView, derived
|
|
5
|
+
# from the system probe's SystemStat (sample[:system]) and reading[:process_rows]. Nil while there
|
|
6
|
+
# is no SystemStat (off macOS, or the probe failed).
|
|
7
|
+
#
|
|
8
|
+
# cpu, user, system percent of the whole machine (0..100): tick deltas over all CPUs since the
|
|
9
|
+
# previous sample / total tick delta. cpu counts user + system + nice
|
|
10
|
+
# ncpu logical CPUs
|
|
11
|
+
# load1/5/15 load averages
|
|
12
|
+
# net_in_rate, bytes/s over the monotonic interval (the probes' at_mono, else the
|
|
13
|
+
# net_out_rate samples' mono)
|
|
14
|
+
# disk_read_rate, bytes/s: the sum of read_rate / write_rate over reading[:process_rows]
|
|
15
|
+
# disk_write_rate where known (readable processes; other users' are not counted). Nil with
|
|
16
|
+
# no rows or no known rate
|
|
17
|
+
# *_trend the last 120 values (four minutes at 2 s), oldest first: metric state
|
|
18
|
+
#
|
|
19
|
+
# Every delta is nil on the first sample and when a counter went backwards (an interface reset,
|
|
20
|
+
# a wrapped tick counter): unknown, never negative. Unknown values are not appended to trends.
|
|
21
|
+
#
|
|
22
|
+
# Metrics are computed on demand (Reading#[]), so reading[:process_rows] is ready whichever order
|
|
23
|
+
# the metrics are registered in. Session focus does not narrow this value: it is machine-wide.
|
|
24
|
+
SystemView = Data.define(
|
|
25
|
+
:cpu, :user, :system, # percent of the machine
|
|
26
|
+
:ncpu,
|
|
27
|
+
:load1, :load5, :load15,
|
|
28
|
+
:net_in_rate, :net_out_rate, # bytes/s
|
|
29
|
+
:disk_read_rate, :disk_write_rate, # bytes/s
|
|
30
|
+
:cpu_trend, :net_in_trend, :net_out_trend, :disk_trend # last TREND values, oldest first
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
module Metrics
|
|
34
|
+
# Named apart from the SystemView/SystemStat models and the system probe's module.
|
|
35
|
+
module MachineLoad
|
|
36
|
+
TREND = 120
|
|
37
|
+
TRENDS = %i[cpu net_in net_out disk].freeze
|
|
38
|
+
|
|
39
|
+
module_function
|
|
40
|
+
|
|
41
|
+
# `now`/`before` are SystemStats (`before` nil on the first sample), `rows` the reading's
|
|
42
|
+
# process rows, `fallback` the samples' monotonic interval; `state` the metric's state Hash.
|
|
43
|
+
def call(now, before, rows, fallback, state)
|
|
44
|
+
return nil unless now
|
|
45
|
+
|
|
46
|
+
user, system, cpu = percents(now.cpu_ticks, before&.cpu_ticks)
|
|
47
|
+
interval = interval(now, before, fallback)
|
|
48
|
+
net_in = rate(now.net_in, before&.net_in, interval)
|
|
49
|
+
net_out = rate(now.net_out, before&.net_out, interval)
|
|
50
|
+
disk_read = disk(rows, :read_rate)
|
|
51
|
+
disk_write = disk(rows, :write_rate)
|
|
52
|
+
disk_total = disk_read && disk_write ? disk_read + disk_write : nil
|
|
53
|
+
|
|
54
|
+
trends = TRENDS.to_h { |name| [name, state[name] ||= []] }
|
|
55
|
+
{ cpu:, net_in:, net_out:, disk: disk_total }.each { |name, value| push(trends[name], value) }
|
|
56
|
+
|
|
57
|
+
load1, load5, load15 = now.load
|
|
58
|
+
SystemView.new(cpu:, user:, system:, ncpu: now.ncpu, load1:, load5:, load15:,
|
|
59
|
+
net_in_rate: net_in, net_out_rate: net_out,
|
|
60
|
+
disk_read_rate: disk_read, disk_write_rate: disk_write,
|
|
61
|
+
cpu_trend: trends[:cpu].dup.freeze, net_in_trend: trends[:net_in].dup.freeze,
|
|
62
|
+
net_out_trend: trends[:net_out].dup.freeze, disk_trend: trends[:disk].dup.freeze)
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
# [user, system, busy] percents between two [user, system, idle, nice] tick readings, or
|
|
66
|
+
# [nil, nil, nil] when either is unknown, no ticks passed, or a counter went backwards.
|
|
67
|
+
def percents(now, was)
|
|
68
|
+
return [nil, nil, nil] unless now && was
|
|
69
|
+
|
|
70
|
+
user, system, idle, nice = now.zip(was).map { |a, b| a - b }
|
|
71
|
+
total = user + system + idle + nice
|
|
72
|
+
return [nil, nil, nil] if [user, system, idle, nice].any?(&:negative?) || !total.positive?
|
|
73
|
+
|
|
74
|
+
pct = ->(ticks) { ticks * 100.0 / total }
|
|
75
|
+
[pct[user], pct[system], pct[user + system + nice]]
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
# Seconds between the two probe reads, else between the samples.
|
|
79
|
+
def interval(now, before, fallback)
|
|
80
|
+
return nil unless before
|
|
81
|
+
|
|
82
|
+
if now.at_mono && before.at_mono
|
|
83
|
+
now.at_mono - before.at_mono
|
|
84
|
+
else
|
|
85
|
+
fallback
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# Bytes/s between two readings of a cumulative counter; nil when unknown or it went backwards.
|
|
90
|
+
def rate(now, was, interval)
|
|
91
|
+
return nil if now.nil? || was.nil? || interval.nil? || !interval.positive? || now < was
|
|
92
|
+
|
|
93
|
+
(now - was) / interval.to_f
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def disk(rows, field)
|
|
97
|
+
known = rows&.filter_map(&field)
|
|
98
|
+
known.nil? || known.empty? ? nil : known.sum.to_f
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
def push(trend, value)
|
|
102
|
+
return if value.nil?
|
|
103
|
+
|
|
104
|
+
trend << value
|
|
105
|
+
trend.shift while trend.size > TREND
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
metric(:system) do |reading, state|
|
|
111
|
+
Metrics::MachineLoad.call(reading.sample[:system], reading.previous&.[](:system), reading[:process_rows],
|
|
112
|
+
reading.interval, state)
|
|
113
|
+
end
|
|
114
|
+
end
|