agentmon 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. checksums.yaml +7 -0
  2. data/CHANGELOG.md +14 -0
  3. data/LICENSE.txt +21 -0
  4. data/README.md +105 -0
  5. data/exe/agentmon +18 -0
  6. data/lib/agentmon/commands/cli_options.rb +21 -0
  7. data/lib/agentmon/commands/footprint.rb +81 -0
  8. data/lib/agentmon/commands/memory.rb +110 -0
  9. data/lib/agentmon/commands/record.rb +118 -0
  10. data/lib/agentmon/commands/report.rb +127 -0
  11. data/lib/agentmon/commands/sessions.rb +98 -0
  12. data/lib/agentmon/commands/top.rb +87 -0
  13. data/lib/agentmon/commands/tree.rb +96 -0
  14. data/lib/agentmon/darwin.rb +235 -0
  15. data/lib/agentmon/engine.rb +115 -0
  16. data/lib/agentmon/focus.rb +88 -0
  17. data/lib/agentmon/metrics/memory.rb +76 -0
  18. data/lib/agentmon/metrics/network.rb +245 -0
  19. data/lib/agentmon/metrics/pressure_drivers.rb +57 -0
  20. data/lib/agentmon/metrics/process_rates.rb +65 -0
  21. data/lib/agentmon/metrics/process_rows.rb +41 -0
  22. data/lib/agentmon/metrics/session_ledger.rb +160 -0
  23. data/lib/agentmon/metrics/session_memory.rb +84 -0
  24. data/lib/agentmon/metrics/session_names.rb +169 -0
  25. data/lib/agentmon/metrics/sessions.rb +80 -0
  26. data/lib/agentmon/metrics/system.rb +114 -0
  27. data/lib/agentmon/model.rb +263 -0
  28. data/lib/agentmon/probes/cwd.rb +32 -0
  29. data/lib/agentmon/probes/memory.rb +139 -0
  30. data/lib/agentmon/probes/network.rb +326 -0
  31. data/lib/agentmon/probes/processes.rb +72 -0
  32. data/lib/agentmon/probes/system.rb +134 -0
  33. data/lib/agentmon/program.rb +55 -0
  34. data/lib/agentmon/reading.rb +83 -0
  35. data/lib/agentmon/recorders/memory.rb +19 -0
  36. data/lib/agentmon/recorders/sessions.rb +40 -0
  37. data/lib/agentmon/registry.rb +155 -0
  38. data/lib/agentmon/sampler.rb +50 -0
  39. data/lib/agentmon/store.rb +86 -0
  40. data/lib/agentmon/ui/connections.rb +75 -0
  41. data/lib/agentmon/ui/interaction.rb +94 -0
  42. data/lib/agentmon/ui/memory.rb +125 -0
  43. data/lib/agentmon/ui/process_actions.rb +49 -0
  44. data/lib/agentmon/ui/process_detail.rb +148 -0
  45. data/lib/agentmon/ui/process_network.rb +13 -0
  46. data/lib/agentmon/ui/process_scopes.rb +32 -0
  47. data/lib/agentmon/ui/process_waits.rb +99 -0
  48. data/lib/agentmon/ui/processes.rb +42 -0
  49. data/lib/agentmon/ui/session_focus.rb +126 -0
  50. data/lib/agentmon/ui/session_memory.rb +78 -0
  51. data/lib/agentmon/ui/sessions.rb +130 -0
  52. data/lib/agentmon/ui/theme.rb +49 -0
  53. data/lib/agentmon/ui.rb +80 -0
  54. data/lib/agentmon/version.rb +5 -0
  55. data/lib/agentmon/views/dense.rb +157 -0
  56. data/lib/agentmon/views/history.rb +62 -0
  57. data/lib/agentmon/views/signals.rb +279 -0
  58. data/lib/agentmon/views/visual.rb +172 -0
  59. data/lib/agentmon/views/widgets.rb +352 -0
  60. data/lib/agentmon/views.rb +875 -0
  61. data/lib/agentmon.rb +38 -0
  62. metadata +135 -0
@@ -0,0 +1,88 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Agentmon
4
+ # Session focus: one agent session the whole dashboard (or a command) narrows to.
5
+ #
6
+ # The Engine holds the focus (`engine.focus = session_id`, `engine.focus = nil` clears it), and
7
+ # while it is set `engine.current` returns a FocusedReading: the same reading with every
8
+ # per-session value narrowed to that session. Panels and commands keep reading
9
+ # `engine.current[:process_rows]` and so on, and show only the focused session's processes,
10
+ # sessions, drivers and memory without knowing focus exists. Machine-wide values (`:memory`,
11
+ # anything not in FILTERS) pass through unchanged.
12
+ #
13
+ # engine.focus = "claude-4242-1790711088"
14
+ # engine.current[:process_rows] # only that session's processes
15
+ # engine.current.focus # => "claude-4242-1790711088" (nil when unfocused)
16
+ # engine.current(focused: false) # the whole machine, e.g. a picker listing every session
17
+ # engine.focused_session # its Session from the ledger, nil when unfocused or gone
18
+ # engine.focus = nil # everything again
19
+ #
20
+ # The filters are per shape in model.rb, so a new per-session shape (a contract change) adds its
21
+ # filter here in the same PR. `reading.sample` is never filtered: code that reads raw probe
22
+ # values sees the whole machine.
23
+ module Focus
24
+ # metric name => ->(value, session_id, unfocused_reading) { narrowed value }. Values are never
25
+ # nil here (nil passes through).
26
+ FILTERS = {
27
+ process_rows: ->(rows, id, _) { rows.select { |r| r.session_id == id } },
28
+ process_rates: lambda do |rates, id, reading|
29
+ by_pid = reading[:sessions]&.by_pid || {}
30
+ rates.select { |pid, _| by_pid[pid] == id }
31
+ end,
32
+ sessions: lambda do |map, id, _|
33
+ SessionMap.new(sessions: map.sessions.select { |s| s.id == id }, by_pid: map.by_pid.select { |_, s| s == id })
34
+ end,
35
+ session_ledger: ->(sessions, id, _) { sessions.select { |s| s.id == id } },
36
+ pressure_drivers: ->(drivers, id, _) { drivers.select { |d| d.session_id == id } },
37
+ session_memory: ->(rows, id, _) { rows.select { |r| r.session_id == id } },
38
+ net_rates: lambda do |rates, id, reading|
39
+ by_pid = reading[:sessions]&.by_pid || {}
40
+ rates.select { |pid, _| by_pid[pid] == id }
41
+ end,
42
+ session_network: ->(rows, id, _) { rows.select { |r| r.session_id == id } },
43
+ connections: ->(rows, id, _) { rows.select { |r| r.session_id == id } }
44
+ }.freeze
45
+
46
+ module_function
47
+
48
+ # `reading` narrowed to `session_id`; the reading itself when nil.
49
+ def apply(reading, session_id) = session_id ? FocusedReading.new(reading, session_id) : reading
50
+
51
+ # Sessions a user's query names (CLI `--session`, a typed filter): an exact id or root pid
52
+ # first, else every session whose label contains it (case-insensitive). [] when none.
53
+ def find(sessions, query)
54
+ query = query.to_s
55
+ exact = sessions.select { |s| s.id == query || s.root_pid.to_s == query }
56
+ return exact unless exact.empty?
57
+
58
+ sessions.select { |s| s.label.downcase.include?(query.downcase) }
59
+ end
60
+ end
61
+
62
+ # A Reading narrowed to one session (see Focus). Every filtered value is computed once, when
63
+ # the Engine builds it under its lock, so feed threads only look values up.
64
+ class FocusedReading
65
+ attr_reader :focus, :unfocused
66
+
67
+ def initialize(reading, session_id)
68
+ @unfocused = reading
69
+ @focus = session_id
70
+ @values = Focus::FILTERS.each_with_object({}) do |(name, filter), values|
71
+ value = reading[name]
72
+ values[name] = value.nil? ? nil : filter.call(value, session_id, reading)
73
+ end
74
+ end
75
+
76
+ def [](name)
77
+ name = name.to_sym
78
+ @values.key?(name) ? @values[name] : @unfocused[name]
79
+ end
80
+
81
+ def sample = @unfocused.sample
82
+ def previous = @unfocused.previous
83
+ def errors = @unfocused.errors
84
+ def at = @unfocused.at
85
+ def interval = @unfocused.interval
86
+ def key?(name) = @unfocused.key?(name)
87
+ end
88
+ end
@@ -0,0 +1,76 @@
1
+ # frozen_string_literal: true
2
+
3
+ # reading[:memory]: the machine's memory as Activity Monitor breaks it down, as a MemoryView
4
+ # (lib/agentmon/model.rb), derived from the memory probe's MemoryStat (sample[:memory]). Nil while
5
+ # there is no MemoryStat (the probe hasn't merged, or failed before its first read).
6
+ #
7
+ # used app + wired + compressed (Activity Monitor's "Memory Used")
8
+ # compression_ratio compressor_stored / compressed: 3.0 means 3 GiB of app memory squeezed into
9
+ # 1 GiB of RAM; 0.0 when nothing is compressed
10
+ # *_rate swap-ins, swap-outs, compressions and decompressions in bytes/s: the counter
11
+ # deltas over the monotonic interval since the previous sample. 0.0 on the first
12
+ # sample (or after one without memory) and when a counter went backwards (it
13
+ # can reset across sleep); never negative
14
+ # pressure 100 - kern.memorystatus_level, in percent (0 = no pressure)
15
+ # pressure_trend the last 60 pressures (two minutes at 2 s), oldest first: metric state
16
+ #
17
+ # Example, 16 GiB Mac under some load:
18
+ #
19
+ # MemoryView(total: 16G, used: 9G, app: 6G, wired: 2G, compressed: 1G, compression_ratio: 3.0,
20
+ # swapin_rate: 2097152.0, ..., pressure: 30.0, pressure_trend: [28.0, 29.0, 30.0])
21
+ #
22
+ # A field the probe couldn't read (nil) stays nil here: its sum, ratio or rate is unknown, not 0.
23
+ module Agentmon
24
+ module Metrics
25
+ # Named apart from the MemoryView/MemoryStat models and the memory probe's module.
26
+ module MemoryBreakdown
27
+ TREND = 60
28
+ COUNTERS = { swapin_rate: :swapins, swapout_rate: :swapouts,
29
+ compression_rate: :compressions, decompression_rate: :decompressions }.freeze
30
+
31
+ module_function
32
+
33
+ # `now` and `before` are MemoryStats (`before` nil on the first sample); `trend` is the
34
+ # metric's state Array, appended to in place.
35
+ def call(now, before, interval, trend)
36
+ return nil unless now
37
+
38
+ pressure = now.memorystatus_level && (100.0 - now.memorystatus_level)
39
+ if pressure
40
+ trend << pressure
41
+ trend.shift while trend.size > TREND
42
+ end
43
+
44
+ MemoryView.new(
45
+ total: now.total, used: sum(now.app, now.wired, now.compressed),
46
+ app: now.app, wired: now.wired, compressed: now.compressed, cached: now.cached, free: now.free,
47
+ swap_used: now.swap_used, swap_total: now.swap_total,
48
+ compression_ratio: ratio(now.compressor_stored, now.compressed),
49
+ **COUNTERS.to_h { |rate, counter| [rate, rate(now.public_send(counter), before&.public_send(counter), interval)] },
50
+ pressure:, pressure_trend: trend.dup.freeze
51
+ )
52
+ end
53
+
54
+ def sum(*parts) = parts.include?(nil) ? nil : parts.sum
55
+
56
+ def ratio(stored, compressed)
57
+ return nil if stored.nil? || compressed.nil?
58
+
59
+ compressed.positive? ? stored.to_f / compressed : 0.0
60
+ end
61
+
62
+ # Bytes/s between two readings of a cumulative counter.
63
+ def rate(now, was, interval)
64
+ return nil if now.nil?
65
+ return 0.0 if was.nil? || interval.nil? || !interval.positive?
66
+
67
+ [now - was, 0].max / interval.to_f
68
+ end
69
+ end
70
+ end
71
+
72
+ metric(:memory) do |reading, state|
73
+ Metrics::MemoryBreakdown.call(reading.sample[:memory], reading.previous&.[](:memory), reading.interval,
74
+ state[:trend] ||= [])
75
+ end
76
+ end
@@ -0,0 +1,245 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "ipaddr"
4
+
5
+ # Per-process and per-session network, from the nettop snapshots in sample[:network]
6
+ # (probes/network.rb). Every metric here is nil while sample[:network] is nil (no snapshot yet,
7
+ # nettop stalled or missing, off macOS): unknown, never 0.
8
+ #
9
+ # reading[:net_rates]: { pid => NetRates or nil }, bytes/s between the previous nettop snapshot and
10
+ # this one, over the snapshots' own at_mono (nettop runs at 1 s, the engine at 2 s). A pid's rate
11
+ # needs the same process as before: the same start_ticks in sample[:processes] and the same
12
+ # nettop name; a reused pid, a first sighting, or a counter that went down is nil. A process in
13
+ # the sample with no nettop line has no sockets: 0.0. When the sample carries the very snapshot
14
+ # the previous one did (nettop hasn't printed since), the previous rates are kept.
15
+ #
16
+ # reading[:session_network]: one SessionNetwork per session in reading[:sessions], each pid counted
17
+ # once (SessionMap#by_pid). Rates sum the members' known rates (members without sockets add
18
+ # 0.0; nil when none is known). bytes_in/bytes_out add each member's positive counter growth while
19
+ # agentmon watched it (a first sighting adds 0, so these are lower bounds). Trends keep the last
20
+ # TREND known rates. connections counts member flows with a concrete remote; remote_hosts counts
21
+ # distinct remote hosts, loopback and link-local left out.
22
+ #
23
+ # Anomaly rule (flags; [] is normal), checked each time a new snapshot arrives:
24
+ # :out_spike out_rate >= SPIKE_FACTOR (8) x the median of the session's previous BASELINE (30)
25
+ # known out_rate values, and >= SPIKE_FLOOR (1 MiB/s), with at least BASELINE_MIN (5)
26
+ # previous values
27
+ # :in_spike the same for in_rate
28
+ # :many_hosts remote_hosts >= MANY_HOSTS (20)
29
+ #
30
+ # reading[:connections]: one ConnectionRow per flow of a process in an agent session with a
31
+ # concrete remote (no `*` wildcard, not Listen), in nettop's order, with per-flow rates matched on
32
+ # (process identity, protocol, local, remote) between snapshots (nil the first time).
33
+ module Agentmon
34
+ module Metrics
35
+ module Network
36
+ TREND = 120
37
+ SPIKE_FACTOR = 8.0
38
+ SPIKE_FLOOR = 1024.0 * 1024 # bytes/s
39
+ BASELINE = 30
40
+ BASELINE_MIN = 5
41
+ MANY_HOSTS = 20
42
+ NO_SOCKETS = NetRates.new(in_rate: 0.0, out_rate: 0.0)
43
+
44
+ module_function
45
+
46
+ # --- reading[:net_rates] ---
47
+
48
+ def rates(snap, processes, state)
49
+ return nil unless snap
50
+
51
+ ids = identities(processes)
52
+ prev = state[:snap]
53
+ if prev && prev.at_mono == snap.at_mono && state[:rates]
54
+ rates = state[:rates].dup
55
+ else
56
+ rates = between(snap, prev, ids, state[:ids] || {})
57
+ state.merge!(snap:, ids:, rates:)
58
+ rates = rates.dup
59
+ end
60
+ ids.each_key { |pid| rates[pid] = NO_SOCKETS unless rates.key?(pid) || snap.processes.key?(pid) }
61
+ rates.freeze
62
+ end
63
+
64
+ def between(snap, prev, ids, prev_ids)
65
+ seconds = prev && (snap.at_mono - prev.at_mono)
66
+ snap.processes.to_h do |pid, now|
67
+ was = prev&.processes&.[](pid)
68
+ same = was && seconds&.positive? && was.name == now.name && ids[pid] == prev_ids[pid]
69
+ next [pid, nil] unless same
70
+
71
+ in_rate = rate(now.bytes_in, was.bytes_in, seconds)
72
+ out_rate = rate(now.bytes_out, was.bytes_out, seconds)
73
+ [pid, in_rate.nil? && out_rate.nil? ? nil : NetRates.new(in_rate:, out_rate:)]
74
+ end
75
+ end
76
+
77
+ # --- reading[:session_network] ---
78
+
79
+ def sessions(snap, map, rates, processes, state)
80
+ return nil unless snap && map
81
+
82
+ rates ||= {}
83
+ ids = identities(processes)
84
+ fresh = state[:at] != snap.at_mono
85
+ state[:at] = snap.at_mono
86
+ members = map.by_pid.each_with_object(Hash.new { |h, k| h[k] = [] }) { |(pid, id), m| m[id] << pid }
87
+ seen = state[:seen] || {}
88
+ next_seen = seen.select { |(pid, ticks, _), _| ids.key?(pid) && ids[pid] == ticks }
89
+ totals = state[:totals] ||= {}
90
+ trends = state[:trends] ||= {}
91
+ spikes = state[:spikes] ||= {}
92
+
93
+ rows = map.sessions.map do |info|
94
+ pids = members[info.id]
95
+ procs = pids.filter_map { |pid| snap.processes[pid] }
96
+ total = totals[info.id] ||= [0, 0]
97
+ count_bytes(procs, ids, seen, next_seen, total) if fresh
98
+ in_rate = sum(pids, :in_rate, rates, snap)
99
+ out_rate = sum(pids, :out_rate, rates, snap)
100
+ trend = trends[info.id] ||= { in: [], out: [] }
101
+ if fresh
102
+ spikes[info.id] = [(:out_spike if spike?(out_rate, trend[:out])),
103
+ (:in_spike if spike?(in_rate, trend[:in]))].compact
104
+ push(trend[:in], in_rate)
105
+ push(trend[:out], out_rate)
106
+ end
107
+ flows = procs.flat_map(&:flows).select(&:remote_host)
108
+ hosts = flows.map(&:remote_host).reject { |h| local_host?(h) }.uniq.size
109
+ flags = [*spikes[info.id], (:many_hosts if hosts >= MANY_HOSTS)].compact
110
+ SessionNetwork.new(session_id: info.id, label: info.label, in_rate:, out_rate:,
111
+ bytes_in: total[0], bytes_out: total[1], in_trend: trend[:in].dup.freeze,
112
+ out_trend: trend[:out].dup.freeze, connections: flows.size, remote_hosts: hosts,
113
+ flags: flags.freeze)
114
+ end
115
+
116
+ state[:seen] = next_seen if fresh
117
+ alive = map.sessions.to_h { |s| [s.id, true] }
118
+ [totals, trends, spikes].each { |h| h.select! { |id, _| alive[id] } }
119
+ rows
120
+ end
121
+
122
+ # Adds each member's counter growth since it was last seen (first sighting: 0) to `total`.
123
+ def count_bytes(procs, ids, seen, next_seen, total)
124
+ procs.each do |np|
125
+ key = [np.pid, ids[np.pid], np.name]
126
+ was = seen[key]
127
+ if was
128
+ total[0] += growth(np.bytes_in, was[0])
129
+ total[1] += growth(np.bytes_out, was[1])
130
+ end
131
+ next_seen[key] = [np.bytes_in || was&.[](0), np.bytes_out || was&.[](1)]
132
+ end
133
+ end
134
+
135
+ # Sum of the members' known rates; a member with no nettop line has no sockets (0.0).
136
+ def sum(pids, field, rates, snap)
137
+ known = pids.filter_map do |pid|
138
+ if rates.key?(pid) then rates[pid]&.public_send(field)
139
+ elsif !snap.processes.key?(pid) then 0.0
140
+ end
141
+ end
142
+ known.empty? ? nil : known.sum.to_f
143
+ end
144
+
145
+ def spike?(value, history)
146
+ return false if value.nil?
147
+
148
+ base = history.last(BASELINE)
149
+ base.size >= BASELINE_MIN && value >= SPIKE_FLOOR && value >= SPIKE_FACTOR * median(base)
150
+ end
151
+
152
+ def median(values)
153
+ sorted = values.sort
154
+ mid = sorted.size / 2
155
+ sorted.size.odd? ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2.0
156
+ end
157
+
158
+ # --- reading[:connections] ---
159
+
160
+ def connections(snap, map, processes, state)
161
+ return nil unless snap && map
162
+
163
+ ids = identities(processes)
164
+ labels = map.sessions.to_h { |s| [s.id, s.label] }
165
+ prev = state[:snap]
166
+ same = prev && prev.at_mono == snap.at_mono
167
+ seconds = prev && !same ? snap.at_mono - prev.at_mono : nil
168
+ prev_bytes = state[:bytes] || {}
169
+ prev_rates = state[:rates] || {}
170
+ bytes = {}
171
+ new_rates = {}
172
+
173
+ rows = snap.processes.flat_map do |pid, np|
174
+ session_id = map.by_pid[pid] or next []
175
+ np.flows.filter_map do |f|
176
+ next unless f.remote_host && f.state != "Listen"
177
+
178
+ key = [pid, ids[pid], np.name, f.protocol, f.local, f.remote]
179
+ bytes[key] = [f.bytes_in, f.bytes_out]
180
+ in_rate, out_rate = new_rates[key] =
181
+ if same
182
+ prev_rates[key] || [nil, nil]
183
+ elsif (was = prev_bytes[key]) && seconds&.positive?
184
+ [rate(f.bytes_in, was[0], seconds), rate(f.bytes_out, was[1], seconds)]
185
+ else
186
+ [nil, nil]
187
+ end
188
+ ConnectionRow.new(pid:, name: np.name, session_id:, session: labels[session_id], protocol: f.protocol,
189
+ local: f.local, remote: f.remote, remote_host: f.remote_host,
190
+ remote_port: f.remote_port, interface: f.interface, state: f.state,
191
+ bytes_in: f.bytes_in, bytes_out: f.bytes_out, in_rate:, out_rate:)
192
+ end
193
+ end
194
+
195
+ state.merge!(snap:, bytes:, rates: new_rates) unless same
196
+ rows
197
+ end
198
+
199
+ # --- shared ---
200
+
201
+ # pid => start_ticks (nil for unreadable processes: then the nettop name has to match alone).
202
+ def identities(processes) = (processes || []).to_h { |p| [p.pid, p.start_ticks] }
203
+
204
+ # Bytes/s between two cumulative counts; nil when either is unknown or it went down.
205
+ def rate(now, was, seconds)
206
+ return nil if now.nil? || was.nil? || now < was
207
+
208
+ (now - was) / seconds.to_f
209
+ end
210
+
211
+ def growth(now, was) = now && was && now > was ? now - was : 0
212
+
213
+ def push(trend, value)
214
+ return if value.nil?
215
+
216
+ trend << value
217
+ trend.shift while trend.size > TREND
218
+ end
219
+
220
+ # Loopback (127/8, ::1, localhost) or link-local (169.254/16, fe80::/10) hosts.
221
+ def local_host?(host)
222
+ address = host.sub(/%.*\z/, "")
223
+ return true if address == "localhost"
224
+
225
+ ip = IPAddr.new(address)
226
+ ip.loopback? || ip.link_local?
227
+ rescue IPAddr::Error
228
+ false
229
+ end
230
+ end
231
+ end
232
+
233
+ metric(:net_rates) do |reading, state|
234
+ Metrics::Network.rates(reading.sample[:network], reading.sample[:processes], state)
235
+ end
236
+
237
+ metric(:session_network) do |reading, state|
238
+ Metrics::Network.sessions(reading.sample[:network], reading[:sessions], reading[:net_rates],
239
+ reading.sample[:processes], state)
240
+ end
241
+
242
+ metric(:connections) do |reading, state|
243
+ Metrics::Network.connections(reading.sample[:network], reading[:sessions], reading.sample[:processes], state)
244
+ end
245
+ end
@@ -0,0 +1,57 @@
1
+ # frozen_string_literal: true
2
+
3
+ # reading[:pressure_drivers]: the alive agent sessions ranked by how much memory they hold, as
4
+ # PressureDriver records (lib/agentmon/model.rb), largest footprint first. For example:
5
+ #
6
+ # claude 4242 · repo 2.1G 26.3% +1.5M/s
7
+ # Claude 59334 900M 11.0% -12K/s
8
+ #
9
+ # - footprint: the session's physical footprint now (bytes; readable members only).
10
+ # - share: footprint as a percent of used memory (reading[:memory].used); nil while memory is
11
+ # unknown (no memory metric, or it failed this sample).
12
+ # - growth_rate: footprint change in bytes per second over the last WINDOW seconds of samples
13
+ # (oldest kept sample to this one, on the monotonic clock); 0.0 until a session has two samples.
14
+ # Negative when it shrinks.
15
+ #
16
+ # Sessions with zero footprint (all members unreadable, or ended) are left out. Stateful: a short
17
+ # footprint history per session id, dropped when the session is no longer alive.
18
+ module Agentmon
19
+ module Metrics
20
+ class Drivers
21
+ WINDOW = 60.0
22
+
23
+ def initialize(state) = @state = state
24
+
25
+ def call(reading)
26
+ history = (@state[:history] ||= {}) # session id => [[mono, footprint], ...], oldest first
27
+ alive = (reading[:session_ledger] || []).select(&:alive?)
28
+ history.select! { |id, _| alive.any? { |s| s.id == id } }
29
+ used = reading[:memory]&.used
30
+ mono = reading.sample.mono
31
+
32
+ alive.filter_map do |session|
33
+ footprint = session.footprint || 0
34
+ growth = growth(history[session.id] ||= [], mono, footprint)
35
+ next unless footprint.positive?
36
+
37
+ PressureDriver.new(session_id: session.id, label: session.label, footprint:,
38
+ share: used&.positive? ? footprint * 100.0 / used : nil, growth_rate: growth)
39
+ end.sort_by { |d| [-d.footprint, d.session_id] }
40
+ end
41
+
42
+ private
43
+
44
+ # Adds this sample to a session's points, forgets those older than WINDOW and returns the
45
+ # rate from the oldest left to this one.
46
+ def growth(points, mono, footprint)
47
+ points.pop if points.last && points.last[0] >= mono # the same sample seen again
48
+ points << [mono, footprint]
49
+ points.shift while mono - points.first[0] > WINDOW
50
+ since, was = points.first
51
+ mono > since ? (footprint - was) / (mono - since) : 0.0
52
+ end
53
+ end
54
+ end
55
+
56
+ metric(:pressure_drivers) { |reading, state| Metrics::Drivers.new(state).call(reading) }
57
+ end
@@ -0,0 +1,65 @@
1
+ # frozen_string_literal: true
2
+
3
+ # reading[:process_rates]: { pid => ProcessRates } between the previous sample and this one.
4
+ #
5
+ # CPU % is the change in CPU seconds over the monotonic interval, x100 (percent of one core, so a
6
+ # process using four cores shows 400%). Disk rates are byte deltas over the interval. A pid only
7
+ # matches the previous sample when its start time matches too, so a reused pid never produces a
8
+ # rate from someone else's counters. Unreadable and newly seen processes fall back to ps's %cpu
9
+ # and nil disk rates.
10
+ #
11
+ # Paging and scheduling rates the same way: pageins, faults, copy-on-write faults and context
12
+ # switches per second, and `run_wait`, the percent of one core the process spent runnable but
13
+ # waiting for a CPU (runnable time grows while a thread is on a CPU too, so CPU time is taken off).
14
+ module Agentmon
15
+ module Metrics
16
+ # Named apart from the ProcessRates model so `ProcessRates.new` below means the model.
17
+ module Rates
18
+ module_function
19
+
20
+ def call(processes, previous, interval)
21
+ before = (previous || []).select(&:readable).to_h { |p| [p.identity, p] }
22
+ (processes || []).to_h do |p|
23
+ q = p.readable && interval&.positive? && before[p.identity]
24
+ [p.pid, q ? delta(p, q, interval) : ProcessRates.new(cpu: p.ps_cpu, read_rate: nil, write_rate: nil)]
25
+ end
26
+ end
27
+
28
+ def delta(now, was, seconds)
29
+ cpu = [now.cpu_time - was.cpu_time, 0].max
30
+ ProcessRates.new(
31
+ cpu: cpu / seconds * 100,
32
+ read_rate: [now.disk_read - was.disk_read, 0].max / seconds,
33
+ write_rate: [now.disk_written - was.disk_written, 0].max / seconds,
34
+ pagein_rate: per_second(now.pageins, was.pageins, seconds),
35
+ fault_rate: per_second(now.faults, was.faults, seconds, wrap: true),
36
+ cow_fault_rate: per_second(now.cow_faults, was.cow_faults, seconds, wrap: true),
37
+ context_switch_rate: per_second(now.context_switches, was.context_switches, seconds, wrap: true),
38
+ run_wait: run_wait(now, was, cpu, seconds)
39
+ )
40
+ end
41
+
42
+ # Events per second between two counts; nil when either is unknown. The task counters are
43
+ # 32 bits in the kernel, so with `wrap:` a count that went down wrapped at 2**32.
44
+ def per_second(now, was, seconds, wrap: false)
45
+ return nil if now.nil? || was.nil?
46
+
47
+ diff = now - was
48
+ diff %= 2**32 if wrap
49
+ [diff, 0].max / seconds.to_f
50
+ end
51
+
52
+ # Runnable time counts time on a CPU too, so waiting for one is runnable growth minus CPU
53
+ # growth, as percent of one core.
54
+ def run_wait(now, was, cpu, seconds)
55
+ return nil if now.runnable_time.nil? || was.runnable_time.nil?
56
+
57
+ [now.runnable_time - was.runnable_time - cpu, 0].max / seconds * 100
58
+ end
59
+ end
60
+ end
61
+
62
+ metric(:process_rates) do |reading|
63
+ Metrics::Rates.call(reading.sample[:processes], reading.previous&.[](:processes), reading.interval)
64
+ end
65
+ end
@@ -0,0 +1,41 @@
1
+ # frozen_string_literal: true
2
+
3
+ # reading[:process_rows]: one ProcessRow per process, joining the probe's counters with rates,
4
+ # working directories and session labels. What the dashboard's process table and the CLI print.
5
+ module Agentmon
6
+ module Metrics
7
+ module ProcessRows
8
+ module_function
9
+
10
+ def call(processes, rates, cwds, sessions, net_rates = nil)
11
+ cwds ||= {}
12
+ rates ||= {} # nil when a metric failed this sample: show what the probe has
13
+ net_rates ||= {} # nil while there is no nettop snapshot: network rates unknown
14
+ sessions ||= SessionMap.new(sessions: [], by_pid: {})
15
+ labels = sessions.sessions.to_h { |s| [s.id, s.label] }
16
+ (processes || []).map do |p|
17
+ rate = rates[p.pid]
18
+ net = net_rates[p.pid]
19
+ session_id = sessions.by_pid[p.pid]
20
+ ProcessRow.new(
21
+ pid: p.pid, ppid: p.ppid, name: p.name, path: p.path, cwd: cwds[p.pid],
22
+ session: session_id && labels[session_id], session_id:,
23
+ cpu: rate&.cpu, footprint: p.footprint, resident: p.resident, peak_footprint: p.peak_footprint,
24
+ read_rate: rate&.read_rate, write_rate: rate&.write_rate,
25
+ cpu_time: p.cpu_time, disk_written: p.disk_written, started_at: p.started_at, readable: p.readable,
26
+ disk_read: p.disk_read, wired: p.wired, pageins: p.pageins, faults: p.faults, cow_faults: p.cow_faults,
27
+ context_switches: p.context_switches, runnable_time: p.runnable_time, threads: p.threads,
28
+ running_threads: p.running_threads, pagein_rate: rate&.pagein_rate, fault_rate: rate&.fault_rate,
29
+ cow_fault_rate: rate&.cow_fault_rate, context_switch_rate: rate&.context_switch_rate, run_wait: rate&.run_wait,
30
+ net_in_rate: net&.in_rate, net_out_rate: net&.out_rate
31
+ )
32
+ end
33
+ end
34
+ end
35
+ end
36
+
37
+ metric(:process_rows) do |reading|
38
+ Metrics::ProcessRows.call(reading.sample[:processes], reading[:process_rates], reading.sample[:cwd], reading[:sessions],
39
+ reading[:net_rates])
40
+ end
41
+ end