agentmon 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +14 -0
- data/LICENSE.txt +21 -0
- data/README.md +105 -0
- data/exe/agentmon +18 -0
- data/lib/agentmon/commands/cli_options.rb +21 -0
- data/lib/agentmon/commands/footprint.rb +81 -0
- data/lib/agentmon/commands/memory.rb +110 -0
- data/lib/agentmon/commands/record.rb +118 -0
- data/lib/agentmon/commands/report.rb +127 -0
- data/lib/agentmon/commands/sessions.rb +98 -0
- data/lib/agentmon/commands/top.rb +87 -0
- data/lib/agentmon/commands/tree.rb +96 -0
- data/lib/agentmon/darwin.rb +235 -0
- data/lib/agentmon/engine.rb +115 -0
- data/lib/agentmon/focus.rb +88 -0
- data/lib/agentmon/metrics/memory.rb +76 -0
- data/lib/agentmon/metrics/network.rb +245 -0
- data/lib/agentmon/metrics/pressure_drivers.rb +57 -0
- data/lib/agentmon/metrics/process_rates.rb +65 -0
- data/lib/agentmon/metrics/process_rows.rb +41 -0
- data/lib/agentmon/metrics/session_ledger.rb +160 -0
- data/lib/agentmon/metrics/session_memory.rb +84 -0
- data/lib/agentmon/metrics/session_names.rb +169 -0
- data/lib/agentmon/metrics/sessions.rb +80 -0
- data/lib/agentmon/metrics/system.rb +114 -0
- data/lib/agentmon/model.rb +263 -0
- data/lib/agentmon/probes/cwd.rb +32 -0
- data/lib/agentmon/probes/memory.rb +139 -0
- data/lib/agentmon/probes/network.rb +326 -0
- data/lib/agentmon/probes/processes.rb +72 -0
- data/lib/agentmon/probes/system.rb +134 -0
- data/lib/agentmon/program.rb +55 -0
- data/lib/agentmon/reading.rb +83 -0
- data/lib/agentmon/recorders/memory.rb +19 -0
- data/lib/agentmon/recorders/sessions.rb +40 -0
- data/lib/agentmon/registry.rb +155 -0
- data/lib/agentmon/sampler.rb +50 -0
- data/lib/agentmon/store.rb +86 -0
- data/lib/agentmon/ui/connections.rb +75 -0
- data/lib/agentmon/ui/interaction.rb +94 -0
- data/lib/agentmon/ui/memory.rb +125 -0
- data/lib/agentmon/ui/process_actions.rb +49 -0
- data/lib/agentmon/ui/process_detail.rb +148 -0
- data/lib/agentmon/ui/process_network.rb +13 -0
- data/lib/agentmon/ui/process_scopes.rb +32 -0
- data/lib/agentmon/ui/process_waits.rb +99 -0
- data/lib/agentmon/ui/processes.rb +42 -0
- data/lib/agentmon/ui/session_focus.rb +126 -0
- data/lib/agentmon/ui/session_memory.rb +78 -0
- data/lib/agentmon/ui/sessions.rb +130 -0
- data/lib/agentmon/ui/theme.rb +49 -0
- data/lib/agentmon/ui.rb +80 -0
- data/lib/agentmon/version.rb +5 -0
- data/lib/agentmon/views/dense.rb +157 -0
- data/lib/agentmon/views/history.rb +62 -0
- data/lib/agentmon/views/signals.rb +279 -0
- data/lib/agentmon/views/visual.rb +172 -0
- data/lib/agentmon/views/widgets.rb +352 -0
- data/lib/agentmon/views.rb +875 -0
- data/lib/agentmon.rb +38 -0
- metadata +135 -0
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Agentmon
|
|
4
|
+
# Session focus: one agent session the whole dashboard (or a command) narrows to.
|
|
5
|
+
#
|
|
6
|
+
# The Engine holds the focus (`engine.focus = session_id`, `engine.focus = nil` clears it), and
|
|
7
|
+
# while it is set `engine.current` returns a FocusedReading: the same reading with every
|
|
8
|
+
# per-session value narrowed to that session. Panels and commands keep reading
|
|
9
|
+
# `engine.current[:process_rows]` and so on, and show only the focused session's processes,
|
|
10
|
+
# sessions, drivers and memory without knowing focus exists. Machine-wide values (`:memory`,
|
|
11
|
+
# anything not in FILTERS) pass through unchanged.
|
|
12
|
+
#
|
|
13
|
+
# engine.focus = "claude-4242-1790711088"
|
|
14
|
+
# engine.current[:process_rows] # only that session's processes
|
|
15
|
+
# engine.current.focus # => "claude-4242-1790711088" (nil when unfocused)
|
|
16
|
+
# engine.current(focused: false) # the whole machine, e.g. a picker listing every session
|
|
17
|
+
# engine.focused_session # its Session from the ledger, nil when unfocused or gone
|
|
18
|
+
# engine.focus = nil # everything again
|
|
19
|
+
#
|
|
20
|
+
# The filters are per shape in model.rb, so a new per-session shape (a contract change) adds its
|
|
21
|
+
# filter here in the same PR. `reading.sample` is never filtered: code that reads raw probe
|
|
22
|
+
# values sees the whole machine.
|
|
23
|
+
module Focus
|
|
24
|
+
# metric name => ->(value, session_id, unfocused_reading) { narrowed value }. Values are never
|
|
25
|
+
# nil here (nil passes through).
|
|
26
|
+
FILTERS = {
|
|
27
|
+
process_rows: ->(rows, id, _) { rows.select { |r| r.session_id == id } },
|
|
28
|
+
process_rates: lambda do |rates, id, reading|
|
|
29
|
+
by_pid = reading[:sessions]&.by_pid || {}
|
|
30
|
+
rates.select { |pid, _| by_pid[pid] == id }
|
|
31
|
+
end,
|
|
32
|
+
sessions: lambda do |map, id, _|
|
|
33
|
+
SessionMap.new(sessions: map.sessions.select { |s| s.id == id }, by_pid: map.by_pid.select { |_, s| s == id })
|
|
34
|
+
end,
|
|
35
|
+
session_ledger: ->(sessions, id, _) { sessions.select { |s| s.id == id } },
|
|
36
|
+
pressure_drivers: ->(drivers, id, _) { drivers.select { |d| d.session_id == id } },
|
|
37
|
+
session_memory: ->(rows, id, _) { rows.select { |r| r.session_id == id } },
|
|
38
|
+
net_rates: lambda do |rates, id, reading|
|
|
39
|
+
by_pid = reading[:sessions]&.by_pid || {}
|
|
40
|
+
rates.select { |pid, _| by_pid[pid] == id }
|
|
41
|
+
end,
|
|
42
|
+
session_network: ->(rows, id, _) { rows.select { |r| r.session_id == id } },
|
|
43
|
+
connections: ->(rows, id, _) { rows.select { |r| r.session_id == id } }
|
|
44
|
+
}.freeze
|
|
45
|
+
|
|
46
|
+
module_function
|
|
47
|
+
|
|
48
|
+
# `reading` narrowed to `session_id`; the reading itself when nil.
|
|
49
|
+
def apply(reading, session_id) = session_id ? FocusedReading.new(reading, session_id) : reading
|
|
50
|
+
|
|
51
|
+
# Sessions a user's query names (CLI `--session`, a typed filter): an exact id or root pid
|
|
52
|
+
# first, else every session whose label contains it (case-insensitive). [] when none.
|
|
53
|
+
def find(sessions, query)
|
|
54
|
+
query = query.to_s
|
|
55
|
+
exact = sessions.select { |s| s.id == query || s.root_pid.to_s == query }
|
|
56
|
+
return exact unless exact.empty?
|
|
57
|
+
|
|
58
|
+
sessions.select { |s| s.label.downcase.include?(query.downcase) }
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# A Reading narrowed to one session (see Focus). Every filtered value is computed once, when
|
|
63
|
+
# the Engine builds it under its lock, so feed threads only look values up.
|
|
64
|
+
class FocusedReading
|
|
65
|
+
attr_reader :focus, :unfocused
|
|
66
|
+
|
|
67
|
+
def initialize(reading, session_id)
|
|
68
|
+
@unfocused = reading
|
|
69
|
+
@focus = session_id
|
|
70
|
+
@values = Focus::FILTERS.each_with_object({}) do |(name, filter), values|
|
|
71
|
+
value = reading[name]
|
|
72
|
+
values[name] = value.nil? ? nil : filter.call(value, session_id, reading)
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
def [](name)
|
|
77
|
+
name = name.to_sym
|
|
78
|
+
@values.key?(name) ? @values[name] : @unfocused[name]
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def sample = @unfocused.sample
|
|
82
|
+
def previous = @unfocused.previous
|
|
83
|
+
def errors = @unfocused.errors
|
|
84
|
+
def at = @unfocused.at
|
|
85
|
+
def interval = @unfocused.interval
|
|
86
|
+
def key?(name) = @unfocused.key?(name)
|
|
87
|
+
end
|
|
88
|
+
end
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# reading[:memory]: the machine's memory as Activity Monitor breaks it down, as a MemoryView
|
|
4
|
+
# (lib/agentmon/model.rb), derived from the memory probe's MemoryStat (sample[:memory]). Nil while
|
|
5
|
+
# there is no MemoryStat (the probe hasn't merged, or failed before its first read).
|
|
6
|
+
#
|
|
7
|
+
# used app + wired + compressed (Activity Monitor's "Memory Used")
|
|
8
|
+
# compression_ratio compressor_stored / compressed: 3.0 means 3 GiB of app memory squeezed into
|
|
9
|
+
# 1 GiB of RAM; 0.0 when nothing is compressed
|
|
10
|
+
# *_rate swap-ins, swap-outs, compressions and decompressions in bytes/s: the counter
|
|
11
|
+
# deltas over the monotonic interval since the previous sample. 0.0 on the first
|
|
12
|
+
# sample (or after one without memory) and when a counter went backwards (it
|
|
13
|
+
# can reset across sleep); never negative
|
|
14
|
+
# pressure 100 - kern.memorystatus_level, in percent (0 = no pressure)
|
|
15
|
+
# pressure_trend the last 60 pressures (two minutes at 2 s), oldest first: metric state
|
|
16
|
+
#
|
|
17
|
+
# Example, 16 GiB Mac under some load:
|
|
18
|
+
#
|
|
19
|
+
# MemoryView(total: 16G, used: 9G, app: 6G, wired: 2G, compressed: 1G, compression_ratio: 3.0,
|
|
20
|
+
# swapin_rate: 2097152.0, ..., pressure: 30.0, pressure_trend: [28.0, 29.0, 30.0])
|
|
21
|
+
#
|
|
22
|
+
# A field the probe couldn't read (nil) stays nil here: its sum, ratio or rate is unknown, not 0.
|
|
23
|
+
module Agentmon
|
|
24
|
+
module Metrics
|
|
25
|
+
# Named apart from the MemoryView/MemoryStat models and the memory probe's module.
|
|
26
|
+
module MemoryBreakdown
|
|
27
|
+
TREND = 60
|
|
28
|
+
COUNTERS = { swapin_rate: :swapins, swapout_rate: :swapouts,
|
|
29
|
+
compression_rate: :compressions, decompression_rate: :decompressions }.freeze
|
|
30
|
+
|
|
31
|
+
module_function
|
|
32
|
+
|
|
33
|
+
# `now` and `before` are MemoryStats (`before` nil on the first sample); `trend` is the
|
|
34
|
+
# metric's state Array, appended to in place.
|
|
35
|
+
def call(now, before, interval, trend)
|
|
36
|
+
return nil unless now
|
|
37
|
+
|
|
38
|
+
pressure = now.memorystatus_level && (100.0 - now.memorystatus_level)
|
|
39
|
+
if pressure
|
|
40
|
+
trend << pressure
|
|
41
|
+
trend.shift while trend.size > TREND
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
MemoryView.new(
|
|
45
|
+
total: now.total, used: sum(now.app, now.wired, now.compressed),
|
|
46
|
+
app: now.app, wired: now.wired, compressed: now.compressed, cached: now.cached, free: now.free,
|
|
47
|
+
swap_used: now.swap_used, swap_total: now.swap_total,
|
|
48
|
+
compression_ratio: ratio(now.compressor_stored, now.compressed),
|
|
49
|
+
**COUNTERS.to_h { |rate, counter| [rate, rate(now.public_send(counter), before&.public_send(counter), interval)] },
|
|
50
|
+
pressure:, pressure_trend: trend.dup.freeze
|
|
51
|
+
)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def sum(*parts) = parts.include?(nil) ? nil : parts.sum
|
|
55
|
+
|
|
56
|
+
def ratio(stored, compressed)
|
|
57
|
+
return nil if stored.nil? || compressed.nil?
|
|
58
|
+
|
|
59
|
+
compressed.positive? ? stored.to_f / compressed : 0.0
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# Bytes/s between two readings of a cumulative counter.
|
|
63
|
+
def rate(now, was, interval)
|
|
64
|
+
return nil if now.nil?
|
|
65
|
+
return 0.0 if was.nil? || interval.nil? || !interval.positive?
|
|
66
|
+
|
|
67
|
+
[now - was, 0].max / interval.to_f
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
metric(:memory) do |reading, state|
|
|
73
|
+
Metrics::MemoryBreakdown.call(reading.sample[:memory], reading.previous&.[](:memory), reading.interval,
|
|
74
|
+
state[:trend] ||= [])
|
|
75
|
+
end
|
|
76
|
+
end
|
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "ipaddr"
|
|
4
|
+
|
|
5
|
+
# Per-process and per-session network, from the nettop snapshots in sample[:network]
|
|
6
|
+
# (probes/network.rb). Every metric here is nil while sample[:network] is nil (no snapshot yet,
|
|
7
|
+
# nettop stalled or missing, off macOS): unknown, never 0.
|
|
8
|
+
#
|
|
9
|
+
# reading[:net_rates]: { pid => NetRates or nil }, bytes/s between the previous nettop snapshot and
|
|
10
|
+
# this one, over the snapshots' own at_mono (nettop runs at 1 s, the engine at 2 s). A pid's rate
|
|
11
|
+
# needs the same process as before: the same start_ticks in sample[:processes] and the same
|
|
12
|
+
# nettop name; a reused pid, a first sighting, or a counter that went down is nil. A process in
|
|
13
|
+
# the sample with no nettop line has no sockets: 0.0. When the sample carries the very snapshot
|
|
14
|
+
# the previous one did (nettop hasn't printed since), the previous rates are kept.
|
|
15
|
+
#
|
|
16
|
+
# reading[:session_network]: one SessionNetwork per session in reading[:sessions], each pid counted
|
|
17
|
+
# once (SessionMap#by_pid). Rates sum the members' known rates (members without sockets add
|
|
18
|
+
# 0.0; nil when none is known). bytes_in/bytes_out add each member's positive counter growth while
|
|
19
|
+
# agentmon watched it (a first sighting adds 0, so these are lower bounds). Trends keep the last
|
|
20
|
+
# TREND known rates. connections counts member flows with a concrete remote; remote_hosts counts
|
|
21
|
+
# distinct remote hosts, loopback and link-local left out.
|
|
22
|
+
#
|
|
23
|
+
# Anomaly rule (flags; [] is normal), checked each time a new snapshot arrives:
|
|
24
|
+
# :out_spike out_rate >= SPIKE_FACTOR (8) x the median of the session's previous BASELINE (30)
|
|
25
|
+
# known out_rate values, and >= SPIKE_FLOOR (1 MiB/s), with at least BASELINE_MIN (5)
|
|
26
|
+
# previous values
|
|
27
|
+
# :in_spike the same for in_rate
|
|
28
|
+
# :many_hosts remote_hosts >= MANY_HOSTS (20)
|
|
29
|
+
#
|
|
30
|
+
# reading[:connections]: one ConnectionRow per flow of a process in an agent session with a
|
|
31
|
+
# concrete remote (no `*` wildcard, not Listen), in nettop's order, with per-flow rates matched on
|
|
32
|
+
# (process identity, protocol, local, remote) between snapshots (nil the first time).
|
|
33
|
+
module Agentmon
|
|
34
|
+
module Metrics
|
|
35
|
+
module Network
|
|
36
|
+
TREND = 120
|
|
37
|
+
SPIKE_FACTOR = 8.0
|
|
38
|
+
SPIKE_FLOOR = 1024.0 * 1024 # bytes/s
|
|
39
|
+
BASELINE = 30
|
|
40
|
+
BASELINE_MIN = 5
|
|
41
|
+
MANY_HOSTS = 20
|
|
42
|
+
NO_SOCKETS = NetRates.new(in_rate: 0.0, out_rate: 0.0)
|
|
43
|
+
|
|
44
|
+
module_function
|
|
45
|
+
|
|
46
|
+
# --- reading[:net_rates] ---
|
|
47
|
+
|
|
48
|
+
def rates(snap, processes, state)
|
|
49
|
+
return nil unless snap
|
|
50
|
+
|
|
51
|
+
ids = identities(processes)
|
|
52
|
+
prev = state[:snap]
|
|
53
|
+
if prev && prev.at_mono == snap.at_mono && state[:rates]
|
|
54
|
+
rates = state[:rates].dup
|
|
55
|
+
else
|
|
56
|
+
rates = between(snap, prev, ids, state[:ids] || {})
|
|
57
|
+
state.merge!(snap:, ids:, rates:)
|
|
58
|
+
rates = rates.dup
|
|
59
|
+
end
|
|
60
|
+
ids.each_key { |pid| rates[pid] = NO_SOCKETS unless rates.key?(pid) || snap.processes.key?(pid) }
|
|
61
|
+
rates.freeze
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def between(snap, prev, ids, prev_ids)
|
|
65
|
+
seconds = prev && (snap.at_mono - prev.at_mono)
|
|
66
|
+
snap.processes.to_h do |pid, now|
|
|
67
|
+
was = prev&.processes&.[](pid)
|
|
68
|
+
same = was && seconds&.positive? && was.name == now.name && ids[pid] == prev_ids[pid]
|
|
69
|
+
next [pid, nil] unless same
|
|
70
|
+
|
|
71
|
+
in_rate = rate(now.bytes_in, was.bytes_in, seconds)
|
|
72
|
+
out_rate = rate(now.bytes_out, was.bytes_out, seconds)
|
|
73
|
+
[pid, in_rate.nil? && out_rate.nil? ? nil : NetRates.new(in_rate:, out_rate:)]
|
|
74
|
+
end
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# --- reading[:session_network] ---
|
|
78
|
+
|
|
79
|
+
def sessions(snap, map, rates, processes, state)
|
|
80
|
+
return nil unless snap && map
|
|
81
|
+
|
|
82
|
+
rates ||= {}
|
|
83
|
+
ids = identities(processes)
|
|
84
|
+
fresh = state[:at] != snap.at_mono
|
|
85
|
+
state[:at] = snap.at_mono
|
|
86
|
+
members = map.by_pid.each_with_object(Hash.new { |h, k| h[k] = [] }) { |(pid, id), m| m[id] << pid }
|
|
87
|
+
seen = state[:seen] || {}
|
|
88
|
+
next_seen = seen.select { |(pid, ticks, _), _| ids.key?(pid) && ids[pid] == ticks }
|
|
89
|
+
totals = state[:totals] ||= {}
|
|
90
|
+
trends = state[:trends] ||= {}
|
|
91
|
+
spikes = state[:spikes] ||= {}
|
|
92
|
+
|
|
93
|
+
rows = map.sessions.map do |info|
|
|
94
|
+
pids = members[info.id]
|
|
95
|
+
procs = pids.filter_map { |pid| snap.processes[pid] }
|
|
96
|
+
total = totals[info.id] ||= [0, 0]
|
|
97
|
+
count_bytes(procs, ids, seen, next_seen, total) if fresh
|
|
98
|
+
in_rate = sum(pids, :in_rate, rates, snap)
|
|
99
|
+
out_rate = sum(pids, :out_rate, rates, snap)
|
|
100
|
+
trend = trends[info.id] ||= { in: [], out: [] }
|
|
101
|
+
if fresh
|
|
102
|
+
spikes[info.id] = [(:out_spike if spike?(out_rate, trend[:out])),
|
|
103
|
+
(:in_spike if spike?(in_rate, trend[:in]))].compact
|
|
104
|
+
push(trend[:in], in_rate)
|
|
105
|
+
push(trend[:out], out_rate)
|
|
106
|
+
end
|
|
107
|
+
flows = procs.flat_map(&:flows).select(&:remote_host)
|
|
108
|
+
hosts = flows.map(&:remote_host).reject { |h| local_host?(h) }.uniq.size
|
|
109
|
+
flags = [*spikes[info.id], (:many_hosts if hosts >= MANY_HOSTS)].compact
|
|
110
|
+
SessionNetwork.new(session_id: info.id, label: info.label, in_rate:, out_rate:,
|
|
111
|
+
bytes_in: total[0], bytes_out: total[1], in_trend: trend[:in].dup.freeze,
|
|
112
|
+
out_trend: trend[:out].dup.freeze, connections: flows.size, remote_hosts: hosts,
|
|
113
|
+
flags: flags.freeze)
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
state[:seen] = next_seen if fresh
|
|
117
|
+
alive = map.sessions.to_h { |s| [s.id, true] }
|
|
118
|
+
[totals, trends, spikes].each { |h| h.select! { |id, _| alive[id] } }
|
|
119
|
+
rows
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
# Adds each member's counter growth since it was last seen (first sighting: 0) to `total`.
|
|
123
|
+
def count_bytes(procs, ids, seen, next_seen, total)
|
|
124
|
+
procs.each do |np|
|
|
125
|
+
key = [np.pid, ids[np.pid], np.name]
|
|
126
|
+
was = seen[key]
|
|
127
|
+
if was
|
|
128
|
+
total[0] += growth(np.bytes_in, was[0])
|
|
129
|
+
total[1] += growth(np.bytes_out, was[1])
|
|
130
|
+
end
|
|
131
|
+
next_seen[key] = [np.bytes_in || was&.[](0), np.bytes_out || was&.[](1)]
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
# Sum of the members' known rates; a member with no nettop line has no sockets (0.0).
|
|
136
|
+
def sum(pids, field, rates, snap)
|
|
137
|
+
known = pids.filter_map do |pid|
|
|
138
|
+
if rates.key?(pid) then rates[pid]&.public_send(field)
|
|
139
|
+
elsif !snap.processes.key?(pid) then 0.0
|
|
140
|
+
end
|
|
141
|
+
end
|
|
142
|
+
known.empty? ? nil : known.sum.to_f
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
def spike?(value, history)
|
|
146
|
+
return false if value.nil?
|
|
147
|
+
|
|
148
|
+
base = history.last(BASELINE)
|
|
149
|
+
base.size >= BASELINE_MIN && value >= SPIKE_FLOOR && value >= SPIKE_FACTOR * median(base)
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
def median(values)
|
|
153
|
+
sorted = values.sort
|
|
154
|
+
mid = sorted.size / 2
|
|
155
|
+
sorted.size.odd? ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2.0
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
# --- reading[:connections] ---
|
|
159
|
+
|
|
160
|
+
def connections(snap, map, processes, state)
|
|
161
|
+
return nil unless snap && map
|
|
162
|
+
|
|
163
|
+
ids = identities(processes)
|
|
164
|
+
labels = map.sessions.to_h { |s| [s.id, s.label] }
|
|
165
|
+
prev = state[:snap]
|
|
166
|
+
same = prev && prev.at_mono == snap.at_mono
|
|
167
|
+
seconds = prev && !same ? snap.at_mono - prev.at_mono : nil
|
|
168
|
+
prev_bytes = state[:bytes] || {}
|
|
169
|
+
prev_rates = state[:rates] || {}
|
|
170
|
+
bytes = {}
|
|
171
|
+
new_rates = {}
|
|
172
|
+
|
|
173
|
+
rows = snap.processes.flat_map do |pid, np|
|
|
174
|
+
session_id = map.by_pid[pid] or next []
|
|
175
|
+
np.flows.filter_map do |f|
|
|
176
|
+
next unless f.remote_host && f.state != "Listen"
|
|
177
|
+
|
|
178
|
+
key = [pid, ids[pid], np.name, f.protocol, f.local, f.remote]
|
|
179
|
+
bytes[key] = [f.bytes_in, f.bytes_out]
|
|
180
|
+
in_rate, out_rate = new_rates[key] =
|
|
181
|
+
if same
|
|
182
|
+
prev_rates[key] || [nil, nil]
|
|
183
|
+
elsif (was = prev_bytes[key]) && seconds&.positive?
|
|
184
|
+
[rate(f.bytes_in, was[0], seconds), rate(f.bytes_out, was[1], seconds)]
|
|
185
|
+
else
|
|
186
|
+
[nil, nil]
|
|
187
|
+
end
|
|
188
|
+
ConnectionRow.new(pid:, name: np.name, session_id:, session: labels[session_id], protocol: f.protocol,
|
|
189
|
+
local: f.local, remote: f.remote, remote_host: f.remote_host,
|
|
190
|
+
remote_port: f.remote_port, interface: f.interface, state: f.state,
|
|
191
|
+
bytes_in: f.bytes_in, bytes_out: f.bytes_out, in_rate:, out_rate:)
|
|
192
|
+
end
|
|
193
|
+
end
|
|
194
|
+
|
|
195
|
+
state.merge!(snap:, bytes:, rates: new_rates) unless same
|
|
196
|
+
rows
|
|
197
|
+
end
|
|
198
|
+
|
|
199
|
+
# --- shared ---
|
|
200
|
+
|
|
201
|
+
# pid => start_ticks (nil for unreadable processes: then the nettop name has to match alone).
|
|
202
|
+
def identities(processes) = (processes || []).to_h { |p| [p.pid, p.start_ticks] }
|
|
203
|
+
|
|
204
|
+
# Bytes/s between two cumulative counts; nil when either is unknown or it went down.
|
|
205
|
+
def rate(now, was, seconds)
|
|
206
|
+
return nil if now.nil? || was.nil? || now < was
|
|
207
|
+
|
|
208
|
+
(now - was) / seconds.to_f
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
def growth(now, was) = now && was && now > was ? now - was : 0
|
|
212
|
+
|
|
213
|
+
def push(trend, value)
|
|
214
|
+
return if value.nil?
|
|
215
|
+
|
|
216
|
+
trend << value
|
|
217
|
+
trend.shift while trend.size > TREND
|
|
218
|
+
end
|
|
219
|
+
|
|
220
|
+
# Loopback (127/8, ::1, localhost) or link-local (169.254/16, fe80::/10) hosts.
|
|
221
|
+
def local_host?(host)
|
|
222
|
+
address = host.sub(/%.*\z/, "")
|
|
223
|
+
return true if address == "localhost"
|
|
224
|
+
|
|
225
|
+
ip = IPAddr.new(address)
|
|
226
|
+
ip.loopback? || ip.link_local?
|
|
227
|
+
rescue IPAddr::Error
|
|
228
|
+
false
|
|
229
|
+
end
|
|
230
|
+
end
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
metric(:net_rates) do |reading, state|
|
|
234
|
+
Metrics::Network.rates(reading.sample[:network], reading.sample[:processes], state)
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
metric(:session_network) do |reading, state|
|
|
238
|
+
Metrics::Network.sessions(reading.sample[:network], reading[:sessions], reading[:net_rates],
|
|
239
|
+
reading.sample[:processes], state)
|
|
240
|
+
end
|
|
241
|
+
|
|
242
|
+
metric(:connections) do |reading, state|
|
|
243
|
+
Metrics::Network.connections(reading.sample[:network], reading[:sessions], reading.sample[:processes], state)
|
|
244
|
+
end
|
|
245
|
+
end
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# reading[:pressure_drivers]: the alive agent sessions ranked by how much memory they hold, as
|
|
4
|
+
# PressureDriver records (lib/agentmon/model.rb), largest footprint first. For example:
|
|
5
|
+
#
|
|
6
|
+
# claude 4242 · repo 2.1G 26.3% +1.5M/s
|
|
7
|
+
# Claude 59334 900M 11.0% -12K/s
|
|
8
|
+
#
|
|
9
|
+
# - footprint: the session's physical footprint now (bytes; readable members only).
|
|
10
|
+
# - share: footprint as a percent of used memory (reading[:memory].used); nil while memory is
|
|
11
|
+
# unknown (no memory metric, or it failed this sample).
|
|
12
|
+
# - growth_rate: footprint change in bytes per second over the last WINDOW seconds of samples
|
|
13
|
+
# (oldest kept sample to this one, on the monotonic clock); 0.0 until a session has two samples.
|
|
14
|
+
# Negative when it shrinks.
|
|
15
|
+
#
|
|
16
|
+
# Sessions with zero footprint (all members unreadable, or ended) are left out. Stateful: a short
|
|
17
|
+
# footprint history per session id, dropped when the session is no longer alive.
|
|
18
|
+
module Agentmon
|
|
19
|
+
module Metrics
|
|
20
|
+
class Drivers
|
|
21
|
+
WINDOW = 60.0
|
|
22
|
+
|
|
23
|
+
def initialize(state) = @state = state
|
|
24
|
+
|
|
25
|
+
def call(reading)
|
|
26
|
+
history = (@state[:history] ||= {}) # session id => [[mono, footprint], ...], oldest first
|
|
27
|
+
alive = (reading[:session_ledger] || []).select(&:alive?)
|
|
28
|
+
history.select! { |id, _| alive.any? { |s| s.id == id } }
|
|
29
|
+
used = reading[:memory]&.used
|
|
30
|
+
mono = reading.sample.mono
|
|
31
|
+
|
|
32
|
+
alive.filter_map do |session|
|
|
33
|
+
footprint = session.footprint || 0
|
|
34
|
+
growth = growth(history[session.id] ||= [], mono, footprint)
|
|
35
|
+
next unless footprint.positive?
|
|
36
|
+
|
|
37
|
+
PressureDriver.new(session_id: session.id, label: session.label, footprint:,
|
|
38
|
+
share: used&.positive? ? footprint * 100.0 / used : nil, growth_rate: growth)
|
|
39
|
+
end.sort_by { |d| [-d.footprint, d.session_id] }
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
private
|
|
43
|
+
|
|
44
|
+
# Adds this sample to a session's points, forgets those older than WINDOW and returns the
|
|
45
|
+
# rate from the oldest left to this one.
|
|
46
|
+
def growth(points, mono, footprint)
|
|
47
|
+
points.pop if points.last && points.last[0] >= mono # the same sample seen again
|
|
48
|
+
points << [mono, footprint]
|
|
49
|
+
points.shift while mono - points.first[0] > WINDOW
|
|
50
|
+
since, was = points.first
|
|
51
|
+
mono > since ? (footprint - was) / (mono - since) : 0.0
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
metric(:pressure_drivers) { |reading, state| Metrics::Drivers.new(state).call(reading) }
|
|
57
|
+
end
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# reading[:process_rates]: { pid => ProcessRates } between the previous sample and this one.
|
|
4
|
+
#
|
|
5
|
+
# CPU % is the change in CPU seconds over the monotonic interval, x100 (percent of one core, so a
|
|
6
|
+
# process using four cores shows 400%). Disk rates are byte deltas over the interval. A pid only
|
|
7
|
+
# matches the previous sample when its start time matches too, so a reused pid never produces a
|
|
8
|
+
# rate from someone else's counters. Unreadable and newly seen processes fall back to ps's %cpu
|
|
9
|
+
# and nil disk rates.
|
|
10
|
+
#
|
|
11
|
+
# Paging and scheduling rates the same way: pageins, faults, copy-on-write faults and context
|
|
12
|
+
# switches per second, and `run_wait`, the percent of one core the process spent runnable but
|
|
13
|
+
# waiting for a CPU (runnable time grows while a thread is on a CPU too, so CPU time is taken off).
|
|
14
|
+
module Agentmon
|
|
15
|
+
module Metrics
|
|
16
|
+
# Named apart from the ProcessRates model so `ProcessRates.new` below means the model.
|
|
17
|
+
module Rates
|
|
18
|
+
module_function
|
|
19
|
+
|
|
20
|
+
def call(processes, previous, interval)
|
|
21
|
+
before = (previous || []).select(&:readable).to_h { |p| [p.identity, p] }
|
|
22
|
+
(processes || []).to_h do |p|
|
|
23
|
+
q = p.readable && interval&.positive? && before[p.identity]
|
|
24
|
+
[p.pid, q ? delta(p, q, interval) : ProcessRates.new(cpu: p.ps_cpu, read_rate: nil, write_rate: nil)]
|
|
25
|
+
end
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def delta(now, was, seconds)
|
|
29
|
+
cpu = [now.cpu_time - was.cpu_time, 0].max
|
|
30
|
+
ProcessRates.new(
|
|
31
|
+
cpu: cpu / seconds * 100,
|
|
32
|
+
read_rate: [now.disk_read - was.disk_read, 0].max / seconds,
|
|
33
|
+
write_rate: [now.disk_written - was.disk_written, 0].max / seconds,
|
|
34
|
+
pagein_rate: per_second(now.pageins, was.pageins, seconds),
|
|
35
|
+
fault_rate: per_second(now.faults, was.faults, seconds, wrap: true),
|
|
36
|
+
cow_fault_rate: per_second(now.cow_faults, was.cow_faults, seconds, wrap: true),
|
|
37
|
+
context_switch_rate: per_second(now.context_switches, was.context_switches, seconds, wrap: true),
|
|
38
|
+
run_wait: run_wait(now, was, cpu, seconds)
|
|
39
|
+
)
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
# Events per second between two counts; nil when either is unknown. The task counters are
|
|
43
|
+
# 32 bits in the kernel, so with `wrap:` a count that went down wrapped at 2**32.
|
|
44
|
+
def per_second(now, was, seconds, wrap: false)
|
|
45
|
+
return nil if now.nil? || was.nil?
|
|
46
|
+
|
|
47
|
+
diff = now - was
|
|
48
|
+
diff %= 2**32 if wrap
|
|
49
|
+
[diff, 0].max / seconds.to_f
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
# Runnable time counts time on a CPU too, so waiting for one is runnable growth minus CPU
|
|
53
|
+
# growth, as percent of one core.
|
|
54
|
+
def run_wait(now, was, cpu, seconds)
|
|
55
|
+
return nil if now.runnable_time.nil? || was.runnable_time.nil?
|
|
56
|
+
|
|
57
|
+
[now.runnable_time - was.runnable_time - cpu, 0].max / seconds * 100
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
metric(:process_rates) do |reading|
|
|
63
|
+
Metrics::Rates.call(reading.sample[:processes], reading.previous&.[](:processes), reading.interval)
|
|
64
|
+
end
|
|
65
|
+
end
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# reading[:process_rows]: one ProcessRow per process, joining the probe's counters with rates,
|
|
4
|
+
# working directories and session labels. What the dashboard's process table and the CLI print.
|
|
5
|
+
module Agentmon
|
|
6
|
+
module Metrics
|
|
7
|
+
module ProcessRows
|
|
8
|
+
module_function
|
|
9
|
+
|
|
10
|
+
def call(processes, rates, cwds, sessions, net_rates = nil)
|
|
11
|
+
cwds ||= {}
|
|
12
|
+
rates ||= {} # nil when a metric failed this sample: show what the probe has
|
|
13
|
+
net_rates ||= {} # nil while there is no nettop snapshot: network rates unknown
|
|
14
|
+
sessions ||= SessionMap.new(sessions: [], by_pid: {})
|
|
15
|
+
labels = sessions.sessions.to_h { |s| [s.id, s.label] }
|
|
16
|
+
(processes || []).map do |p|
|
|
17
|
+
rate = rates[p.pid]
|
|
18
|
+
net = net_rates[p.pid]
|
|
19
|
+
session_id = sessions.by_pid[p.pid]
|
|
20
|
+
ProcessRow.new(
|
|
21
|
+
pid: p.pid, ppid: p.ppid, name: p.name, path: p.path, cwd: cwds[p.pid],
|
|
22
|
+
session: session_id && labels[session_id], session_id:,
|
|
23
|
+
cpu: rate&.cpu, footprint: p.footprint, resident: p.resident, peak_footprint: p.peak_footprint,
|
|
24
|
+
read_rate: rate&.read_rate, write_rate: rate&.write_rate,
|
|
25
|
+
cpu_time: p.cpu_time, disk_written: p.disk_written, started_at: p.started_at, readable: p.readable,
|
|
26
|
+
disk_read: p.disk_read, wired: p.wired, pageins: p.pageins, faults: p.faults, cow_faults: p.cow_faults,
|
|
27
|
+
context_switches: p.context_switches, runnable_time: p.runnable_time, threads: p.threads,
|
|
28
|
+
running_threads: p.running_threads, pagein_rate: rate&.pagein_rate, fault_rate: rate&.fault_rate,
|
|
29
|
+
cow_fault_rate: rate&.cow_fault_rate, context_switch_rate: rate&.context_switch_rate, run_wait: rate&.run_wait,
|
|
30
|
+
net_in_rate: net&.in_rate, net_out_rate: net&.out_rate
|
|
31
|
+
)
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
metric(:process_rows) do |reading|
|
|
38
|
+
Metrics::ProcessRows.call(reading.sample[:processes], reading[:process_rates], reading.sample[:cwd], reading[:sessions],
|
|
39
|
+
reading[:net_rates])
|
|
40
|
+
end
|
|
41
|
+
end
|