alissa-tools-github-revloop 0.16.14__tar.gz → 0.16.16__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. {alissa_tools_github_revloop-0.16.14/src/main/alissa_tools_github_revloop.egg-info → alissa_tools_github_revloop-0.16.16}/PKG-INFO +1 -1
  2. alissa_tools_github_revloop-0.16.16/src/main/alissa/tools/github/revloop/version +1 -0
  3. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/webui/__init__.py +7 -5
  4. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/webui/page.py +59 -1
  5. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/webui/sources.py +47 -6
  6. alissa_tools_github_revloop-0.16.16/src/main/alissa/tools/github/revloop/webui/sysinfo.py +356 -0
  7. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16/src/main/alissa_tools_github_revloop.egg-info}/PKG-INFO +1 -1
  8. alissa_tools_github_revloop-0.16.14/src/main/alissa/tools/github/revloop/version +0 -1
  9. alissa_tools_github_revloop-0.16.14/src/main/alissa/tools/github/revloop/webui/sysinfo.py +0 -174
  10. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/LICENSE +0 -0
  11. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/MANIFEST.in +0 -0
  12. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/NOTICE +0 -0
  13. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/README.md +0 -0
  14. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/requirements.txt +0 -0
  15. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/setup.cfg +0 -0
  16. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/setup.py +0 -0
  17. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/__init__.py +0 -0
  18. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/__main__.py +0 -0
  19. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/alissa.py +0 -0
  20. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/config.py +0 -0
  21. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/ghclient.py +0 -0
  22. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/loop.py +0 -0
  23. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/proc.py +0 -0
  24. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/prreview.py +0 -0
  25. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/state.py +0 -0
  26. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/version.py +0 -0
  27. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/webui/__main__.py +0 -0
  28. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/webui/auth.py +0 -0
  29. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa/tools/github/revloop/webui/server.py +0 -0
  30. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa_tools_github_revloop.egg-info/SOURCES.txt +0 -0
  31. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa_tools_github_revloop.egg-info/dependency_links.txt +0 -0
  32. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa_tools_github_revloop.egg-info/entry_points.txt +0 -0
  33. {alissa_tools_github_revloop-0.16.14 → alissa_tools_github_revloop-0.16.16}/src/main/alissa_tools_github_revloop.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: alissa-tools-github-revloop
3
- Version: 0.16.14
3
+ Version: 0.16.16
4
4
  Summary: ALISSA-TOOLS-GITHUB-REVLOOP
5
5
  Home-page: https://alissa.app
6
6
  Author: Fahera
@@ -15,14 +15,16 @@ pass writes one `poll_snapshots` row (UI-1, PR #35) carrying the pass timing,
15
15
  the candidate count, the decision-summary counts, and the compact per-item
16
16
  stage list. The sidecar reads that table through `State.read_snapshots`, plus
17
17
  the spawn ledger, the escalation table and the ping ledger (the operator
18
- inbox), all read-only. Its only live signals are local (`alissa tmux ls`, a
19
- `/proc` walk of each session's pane-PID tree) or cached (`gh api rate_limit`,
20
- 60s; the PyPI version JSON, 10m) -- so a fleet of operators refreshing the
21
- dashboard never moves the daemon's rate budget.
18
+ inbox), all read-only. Its only live signals are local (`alissa tmux ls`, one
19
+ `/proc` walk serving both each session's pane-PID tree and the host-wide
20
+ top-by-RSS list, and the container's own cgroup v2 memory charge) or cached
21
+ (`gh api rate_limit`, 60s; the PyPI version JSON, 10m) -- so a fleet of
22
+ operators refreshing the dashboard never moves the daemon's rate budget.
22
23
 
23
24
  Layout:
24
25
  auth.py -- fail-closed passcode, HMAC-signed sessions, CSRF, login throttle
25
- sysinfo.py -- /proc process-tree CPU%/RSS (sample-free, vanished-PID tolerant)
26
+ sysinfo.py -- /proc process-tree CPU%/RSS + host-wide top-by-RSS and the
27
+ cgroup memory split (sample-free, vanished-PID tolerant)
26
28
  sources.py -- the read-only data layer + the retry-now UPDATE, cached checks
27
29
  page.py -- the single static HTML page (studio design system, both themes)
28
30
  server.py -- ThreadingHTTPServer wiring, routing, auth/CSRF gating, actions
@@ -165,7 +165,7 @@ header.top h1 {
165
165
 
166
166
  /* stat tiles: seamless 1px-gap grid */
167
167
  .tiles {
168
- display: grid; grid-template-columns: repeat(4, 1fr); gap: 1px;
168
+ display: grid; grid-template-columns: repeat(5, 1fr); gap: 1px;
169
169
  background: var(--surface-border); border: 1px solid var(--surface-border);
170
170
  border-radius: var(--radius-lg); overflow: hidden; margin-bottom: 2rem;
171
171
  }
@@ -182,6 +182,9 @@ header.top h1 {
182
182
  .meter.crit > span { background: var(--status-cancelled); }
183
183
 
184
184
  .grid2 { display: grid; grid-template-columns: 1fr 1fr; gap: 1.5rem; }
185
+ /* Five tiles do not fit a laptop viewport at the tile's own type scale; the
186
+ 900px rule below must stay LAST so the narrow case still wins. */
187
+ @media (max-width: 1100px) { .tiles { grid-template-columns: repeat(3, 1fr); } }
185
188
  @media (max-width: 900px) { .grid2 { grid-template-columns: 1fr; }
186
189
  .tiles { grid-template-columns: repeat(2, 1fr); } }
187
190
 
@@ -331,6 +334,11 @@ _DASHBOARD = """<!doctype html>
331
334
 
332
335
  <section class="panel"><p class="overline">Sessions</p><div id="sessions"></div></section>
333
336
 
337
+ <section class="panel">
338
+ <p class="overline">Top Processes &middot; by RSS, host-wide</p>
339
+ <div id="topprocs"></div>
340
+ </section>
341
+
334
342
  <section class="panel">
335
343
  <p class="overline">Daemon Log &middot; <span id="log-path" class="mono muted"></span></p>
336
344
  <div class="log" id="log"></div>
@@ -422,6 +430,36 @@ _JS = r"""
422
430
  var vsub = bytes(t.volume.used_bytes) + ' / ' + bytes(t.volume.total_bytes);
423
431
  out += tile('Volume', t.volume.percent + '%', vsub, t.volume.percent);
424
432
  } else { out += tile('Volume', '--', 'unavailable'); }
433
+ // The tile a platform memory graph sends you here to read: the headline is
434
+ // what the container is CHARGED, the sub splits it into what is really
435
+ // process memory and what is cache the kernel would drop under pressure --
436
+ // so a "6 GB" plateau reads as "72 MB real + 5.9 GB cache" at a glance.
437
+ // The meter is the RESIDENT share of the charge, deliberately: it is the
438
+ // only part of a rising charge that a limit can actually kill for, so the
439
+ // shared warn/crit thresholds mean the same thing here as on the others.
440
+ // Gated on ANY of the three, not on `charged`: cgroup_memory reads
441
+ // memory.current and memory.stat through separate helpers so each degrades
442
+ // on its own, and a tile keyed to the headline alone would throw away a
443
+ // breakdown the reader deliberately preserved. The headline and the meter
444
+ // carry their own null handling, so a missing `charged` costs only itself.
445
+ var mem = t.memory;
446
+ if (mem && (mem.charged != null || mem.resident != null || mem.reclaimable != null)) {
447
+ var msub = (mem.resident == null ? '--' : bytes(mem.resident)) + ' resident · ' +
448
+ (mem.reclaimable == null ? '--' : bytes(mem.reclaimable)) + ' reclaimable';
449
+ // shmem is charged, is NOT droppable (swap-backed), and is excluded from
450
+ // reclaimable -- so it has to be visible, or a tmpfs-heavy container
451
+ // shows a charge that neither of the other two numbers accounts for.
452
+ if (mem.shmem) msub += ' · ' + bytes(mem.shmem) + ' shmem';
453
+ var mpct = (mem.resident == null || !mem.charged) ? null :
454
+ Math.round(100 * mem.resident / mem.charged);
455
+ out += tile('Container Memory', bytes(mem.charged), msub, mpct);
456
+ // 'unavailable', not a cgroup-v2 verdict: this branch is also reached on a
457
+ // host-namespace deployment (the controller exists, its interface files
458
+ // are absent from the v2 root) and on a partial read. The console cannot
459
+ // tell those apart, and a wrong diagnosis sends an operator to check the
460
+ // wrong thing -- the neighbouring tiles say 'unavailable' for the same
461
+ // reason, and so does the origin task's acceptance detail.
462
+ } else { out += tile('Container Memory', '--', 'unavailable'); }
425
463
  out += tile('Review Queue', t.queue_depth, 'PRs awaiting me, last poll');
426
464
  el('tiles').innerHTML = out;
427
465
  }
@@ -524,6 +562,25 @@ _JS = r"""
524
562
  });
525
563
  }
526
564
 
565
+ // Host-wide, not per session: it names whatever holds a resident charge,
566
+ // which is routinely not a reviewer at all. Read-only -- no kill button
567
+ // here, because these PIDs are unmanaged and the sessions panel above is
568
+ // the only place a process should be killed from.
569
+ function renderTopProcs(rows) {
570
+ rows = rows || [];
571
+ if (!rows.length) {
572
+ el('topprocs').innerHTML = '<div class="empty">No process data (/proc unreadable).</div>';
573
+ return;
574
+ }
575
+ var head = '<table><thead><tr><th class="num">PID</th><th>Process</th>' +
576
+ '<th class="num">RSS</th></tr></thead><tbody>';
577
+ el('topprocs').innerHTML = head + rows.map(function (p) {
578
+ return '<tr><td class="num mono">' + esc(p.pid) + '</td>' +
579
+ '<td class="mono">' + esc(p.comm) + '</td>' +
580
+ '<td class="num">' + bytes(p.rss_bytes) + '</td></tr>';
581
+ }).join('') + '</tbody></table>';
582
+ }
583
+
527
584
  function renderLog(log) {
528
585
  el('log-path').textContent = log.path || '(no log configured)';
529
586
  el('log').textContent = log.lines.length ? log.lines.join('\n') : '(log empty or unavailable)';
@@ -558,6 +615,7 @@ _JS = r"""
558
615
  renderPipeline(d.pipeline);
559
616
  renderInbox(d.inbox);
560
617
  renderSessions(d.sessions);
618
+ renderTopProcs(d.top_procs);
561
619
  renderLog(d.log);
562
620
  var when = new Date(d.generated_at * 1000).toLocaleTimeString();
563
621
  el('status-line').textContent = 'updated ' + when;
@@ -7,7 +7,10 @@ in strict budget order:
7
7
  UI-1 reader), the spawn ledger, the escalation table and the ping ledger. No
8
8
  GitHub call: the daemon already wrote everything down.
9
9
  2. **Local process state** -- `alissa tmux ls --json` for the session list, and
10
- a `/proc` walk (sysinfo) of each session's pane-PID tree for CPU%/RSS.
10
+ a `/proc` walk (sysinfo) of each session's pane-PID tree for CPU%/RSS, plus
11
+ the two container-wide reads the per-session sums cannot answer: the cgroup
12
+ memory charge split into resident vs reclaimable, and the top processes by
13
+ RSS across the whole host. One `/proc` scan serves both.
11
14
  3. **Two cached remote checks** -- `gh api rate_limit` (60s cache) for the rate
12
15
  meter, and the PyPI version JSON (10m cache) for the running-vs-latest drift
13
16
  chip. These are the *only* network calls, and both are cached so a room full
@@ -61,6 +64,10 @@ from . import sysinfo
61
64
 
62
65
  # How many recent snapshots feed the sparklines / pipeline board.
63
66
  SPARK_POINTS = 60
67
+ # How many processes the host-wide top-by-RSS list carries. Five is the whole
68
+ # point of the panel: it names what is holding a resident charge, it is not a
69
+ # process browser, and every extra row is payload on a ~10s poll.
70
+ TOP_PROCS = 5
64
71
  # Cache lifetimes for the remote checks (seconds).
65
72
  RATE_CACHE_TTL = 60.0
66
73
  VERSION_CACHE_TTL = 600.0
@@ -156,6 +163,7 @@ class Sources:
156
163
  run: "Callable[..., str]" = proc_run,
157
164
  http_get: "Callable[[str, float], bytes | None]" = _default_http_get,
158
165
  proc_root: str = "/proc",
166
+ cgroup_root: str = "/sys/fs/cgroup",
159
167
  clock: Callable[[], float] = time.monotonic,
160
168
  wall_clock: Callable[[], float] = time.time,
161
169
  ) -> None:
@@ -165,6 +173,7 @@ class Sources:
165
173
  self._run = run
166
174
  self._http_get = http_get
167
175
  self._proc_root = proc_root
176
+ self._cgroup_root = cgroup_root
168
177
  self._clock = clock
169
178
  self._wall = wall_clock
170
179
  self._rate_cache = _Cache(RATE_CACHE_TTL, clock)
@@ -283,7 +292,11 @@ class Sources:
283
292
  rows = self._read_state([], lambda st: st.read_spawns(sessions=names))
284
293
  return {row["session"]: row for row in rows}
285
294
 
286
- def sessions(self, spawns: "list[dict] | None" = None) -> "list[dict]":
295
+ def sessions(
296
+ self,
297
+ spawns: "list[dict] | None" = None,
298
+ index: "tuple[dict[int, list[int]], dict[int, dict]] | None" = None,
299
+ ) -> "list[dict]":
287
300
  """The managed-session table: liveness from `alissa tmux ls`, footprint
288
301
  from /proc, and the PR round each session is reviewing.
289
302
 
@@ -296,6 +309,12 @@ class Sources:
296
309
  `spawns` supplies the ledger rows directly; when it is None (the
297
310
  dashboard's path) they are read here, keyed by the names tmux just
298
311
  returned -- which is why the session list is fetched first.
312
+
313
+ `index` supplies the `/proc` snapshot. The dashboard now needs one
314
+ anyway for the host-wide top-process list, so it builds the index once
315
+ and hands the same one here; passed None (a caller that only wants the
316
+ table) the old lazy build is unchanged and a table with no live pane
317
+ still never scans `/proc`.
299
318
  """
300
319
  raw = self._safe_json(["alissa", "tmux", "ls", "--json"]) or []
301
320
  if not isinstance(raw, list):
@@ -311,9 +330,8 @@ class Sources:
311
330
  out: list[dict] = []
312
331
  # ONE /proc snapshot for the whole table: the index is identical for
313
332
  # every session in this build, so rebuilding it per session would make
314
- # the walk O(sessions x processes). Built lazily -- a table with no
315
- # live pane never scans /proc at all.
316
- index: "tuple[dict[int, list[int]], dict[int, dict]] | None" = None
333
+ # the walk O(sessions x processes). Built lazily when the caller did
334
+ # not supply one -- a table with no live pane never scans /proc at all.
317
335
  for entry in raw:
318
336
  if not isinstance(entry, dict):
319
337
  continue
@@ -479,9 +497,24 @@ class Sources:
479
497
  snaps = self.snapshots(SPARK_POINTS)
480
498
  latest = snaps[0] if snaps else None
481
499
  ledgers = self.ledgers()
482
- sessions = self.sessions()
500
+ # ONE /proc scan for this whole build: the session table walks pane
501
+ # trees out of it and the top-process list ranks the same snapshot, so
502
+ # the two panels can never disagree about a process that exited between
503
+ # them -- and the host pays for one walk per poll, not two.
504
+ #
505
+ # The cost on the other side of that trade, so nobody reorders this
506
+ # thinking it is free: the snapshot is now taken BEFORE `sessions()`
507
+ # shells out to tmux, so a pane that starts inside that window is
508
+ # missing from `stats` and its row shows "--" for CPU%/RSS for one
509
+ # poll. Self-healing in ~10s, and cheaper than walking /proc twice.
510
+ proc_index = sysinfo.build_index(self._proc_root)
511
+ sessions = self.sessions(index=proc_index)
483
512
  rate = self.rate_limit()
484
513
  disk = sysinfo.disk_usage(self.config.workspace_root)
514
+ memory = sysinfo.cgroup_memory(self._cgroup_root)
515
+ top_procs = sysinfo.top_procs(
516
+ TOP_PROCS, proc_root=self._proc_root, index=proc_index
517
+ )
485
518
 
486
519
  # Sparklines want oldest -> newest for left-to-right drawing. "Active"
487
520
  # counts both buckets a live reviewer session sits in: a round enqueued
@@ -522,6 +555,11 @@ class Sources:
522
555
  "live_sessions": sum(1 for s in sessions if s["live"]),
523
556
  "rate": rate,
524
557
  "volume": disk,
558
+ # The container's own charge, split three ways. Every field is
559
+ # None on a host without cgroup v2 (dev laptop, macOS) and the
560
+ # tile renders "unavailable" -- the console must not require
561
+ # Linux to load.
562
+ "memory": memory,
525
563
  "queue_depth": latest["candidates"] if latest else 0,
526
564
  },
527
565
  "sparklines": sparklines,
@@ -533,6 +571,9 @@ class Sources:
533
571
  },
534
572
  "inbox": self._inbox(ledgers["escalations"], ledgers["pings"]),
535
573
  "sessions": sessions,
574
+ # Host-wide, not per session: when the memory tile says the charge
575
+ # IS resident, this is what names the holder.
576
+ "top_procs": top_procs,
536
577
  "log": self.log_tail(),
537
578
  }
538
579
 
@@ -0,0 +1,356 @@
1
+ """Per-session resource accounting off `/proc`, plus workspace disk and
2
+ container-memory usage.
3
+
4
+ The sessions panel wants CPU% and RSS for each managed tmux session. tmux hands
5
+ us a pane PID; the real work runs in that pane's child tree (a shell, the agent,
6
+ its subprocesses), so we sum over the whole tree rooted at the pane PID.
7
+
8
+ Two deliberate properties:
9
+
10
+ * **Sample-free CPU.** Instantaneous CPU% needs two `/proc` reads spaced apart;
11
+ a dashboard endpoint polled every ~10s should not block sampling. Instead we
12
+ report the *cumulative average* since the pane started:
13
+ `100 * (utime+stime)/CLK_TCK / (uptime - starttime/CLK_TCK)`. It answers "how
14
+ hot has this session run over its life", which is the operator question, and
15
+ needs a single read.
16
+ * **Vanished-PID tolerant.** Between listing `/proc` and reading a PID's `stat`,
17
+ the process can exit. Every read that fails (gone, permission, malformed) is
18
+ skipped -- a disappearing reviewer contributes nothing and never raises.
19
+
20
+ The `/proc` root is a parameter so the parsing is testable against a crafted
21
+ fixture directory (including a PID that is indexed but whose `stat` was deleted
22
+ mid-walk -- the vanished-PID case the mutation bar pins).
23
+
24
+ Per-session accounting cannot answer the question a platform memory graph
25
+ asks. A Railway-style flat multi-GB plateau with zero reviewer sessions is
26
+ CONTAINER-wide, and the per-pane sums (tens of MB) say nothing about it, so
27
+ `cgroup_memory` reads the container's own cgroup v2 charge and splits it into
28
+ the three numbers that settle "is that a leak?": what the container is charged
29
+ for, what is really process memory, and what is page cache the kernel would
30
+ drop the moment anything asked for it. `top_procs` is the same question one
31
+ level down -- the whole `/proc`, not one pane tree -- so a charge that IS
32
+ resident can be attributed to a process without shelling into the container.
33
+ Both follow the module's conventions: roots are parameters, every read
34
+ degrades to None instead of raising (a dev laptop or macOS has no cgroup v2
35
+ and must still render the console), and neither samples.
36
+ """
37
+
38
+ from __future__ import annotations
39
+
40
+ import os
41
+ import shutil
42
+ from pathlib import Path
43
+
44
+ _CLK_TCK = os.sysconf("SC_CLK_TCK") if hasattr(os, "sysconf") else 100
45
+ _PAGE_SIZE = os.sysconf("SC_PAGE_SIZE") if hasattr(os, "sysconf") else 4096
46
+
47
+ # Field offsets INTO the tail of /proc/<pid>/stat (everything after the
48
+ # "(comm)" field, which itself is field 3 == index 0 of the tail). See
49
+ # proc(5): utime=14, stime=15, starttime=22, rss=24 -> tail indices below.
50
+ _TAIL_PPID = 1
51
+ _TAIL_UTIME = 11
52
+ _TAIL_STIME = 12
53
+ _TAIL_STARTTIME = 19
54
+ _TAIL_RSS = 21
55
+ _TAIL_MIN_LEN = 22
56
+
57
+ # The cgroup v2 files the memory tile reads, and the `memory.stat` keys it
58
+ # keeps. Deliberately a fixed, small set: `memory.stat` carries ~40 keys and
59
+ # the tile answers one question, so parsing the rest would be payload we never
60
+ # render. `anon` is process memory, `file` is page cache, `slab_reclaimable` is
61
+ # kernel cache the shrinkers can free -- together they are the split between a
62
+ # real footprint and a charge the kernel would give back under pressure.
63
+ _CGROUP_CURRENT = "memory.current"
64
+ _CGROUP_STAT = "memory.stat"
65
+ _CGROUP_STAT_KEYS = (
66
+ "anon",
67
+ "file",
68
+ "inactive_file",
69
+ "shmem",
70
+ "slab_reclaimable",
71
+ "slab_unreclaimable",
72
+ )
73
+
74
+
75
+ def read_stat(proc_root: "str | os.PathLike[str]", pid: int) -> "dict | None":
76
+ """Parse one `/proc/<pid>/stat`, or None if the PID has vanished or the
77
+ line is malformed. The `comm` field can contain spaces and parentheses, so
78
+ we split on the LAST ')' before tokenising -- the canonical safe parse.
79
+
80
+ `comm` is carried in the row (between the FIRST '(' and that last ')', the
81
+ mirror of the same parse) purely for the top-process list: the tree sums
82
+ need only the numbers, but a top-5 by RSS that shows bare PIDs tells an
83
+ operator nothing about what is holding the memory.
84
+ """
85
+ try:
86
+ data = (Path(proc_root) / str(pid) / "stat").read_text()
87
+ except (OSError, ValueError):
88
+ return None
89
+ rparen = data.rfind(")")
90
+ lparen = data.find("(")
91
+ if rparen == -1 or lparen == -1 or lparen > rparen:
92
+ return None
93
+ tail = data[rparen + 1:].split()
94
+ if len(tail) < _TAIL_MIN_LEN:
95
+ return None
96
+ try:
97
+ return {
98
+ "pid": int(pid),
99
+ "comm": data[lparen + 1:rparen],
100
+ "ppid": int(tail[_TAIL_PPID]),
101
+ "cpu_ticks": int(tail[_TAIL_UTIME]) + int(tail[_TAIL_STIME]),
102
+ "starttime": int(tail[_TAIL_STARTTIME]),
103
+ "rss_pages": int(tail[_TAIL_RSS]),
104
+ }
105
+ except (ValueError, IndexError):
106
+ return None
107
+
108
+
109
+ def read_uptime(proc_root: "str | os.PathLike[str]") -> "float | None":
110
+ try:
111
+ return float((Path(proc_root) / "uptime").read_text().split()[0])
112
+ except (OSError, ValueError, IndexError):
113
+ return None
114
+
115
+
116
+ def build_index(
117
+ proc_root: "str | os.PathLike[str]",
118
+ ) -> "tuple[dict[int, list[int]], dict[int, dict]]":
119
+ """Scan `/proc` once: return (ppid -> [child pids], pid -> stat). Numeric
120
+ entries only; anything that vanishes mid-scan is silently skipped."""
121
+ children: dict[int, list[int]] = {}
122
+ stats: dict[int, dict] = {}
123
+ try:
124
+ entries = os.listdir(proc_root)
125
+ except OSError:
126
+ return children, stats
127
+ for name in entries:
128
+ if not name.isdigit():
129
+ continue
130
+ st = read_stat(proc_root, int(name))
131
+ if st is None:
132
+ continue
133
+ stats[st["pid"]] = st
134
+ children.setdefault(st["ppid"], []).append(st["pid"])
135
+ return children, stats
136
+
137
+
138
+ def tree_pids(root_pid: int, children: "dict[int, list[int]]") -> "set[int]":
139
+ """Every PID in the subtree rooted at `root_pid` (inclusive). Iterative and
140
+ cycle-guarded -- a corrupt ppid loop can never spin forever."""
141
+ seen: set[int] = set()
142
+ stack = [root_pid]
143
+ while stack:
144
+ pid = stack.pop()
145
+ if pid in seen:
146
+ continue
147
+ seen.add(pid)
148
+ stack.extend(children.get(pid, ()))
149
+ return seen
150
+
151
+
152
+ def tree_usage(
153
+ root_pid: "int | None",
154
+ *,
155
+ proc_root: "str | os.PathLike[str]" = "/proc",
156
+ index: "tuple[dict[int, list[int]], dict[int, dict]] | None" = None,
157
+ ) -> dict:
158
+ """CPU% (cumulative average since the pane started) and summed RSS for the
159
+ whole process tree under `root_pid`. Returns None/None when the root has
160
+ already vanished -- "unknown", the same thing the caller shows for a
161
+ session with no pane at all, so the column reads consistently.
162
+
163
+ `index` is the (children, stats) pair from `build_index`. Pass it when
164
+ accounting SEVERAL trees from one snapshot of `/proc`: the index is
165
+ identical for every session in one dashboard build, and rebuilding it per
166
+ session makes the walk O(sessions x processes) -- on a busy reviewer host
167
+ thousands of `stat` reads every poll, stolen from the reviewer agents the
168
+ console exists to watch. Omitted, it is built once for this call.
169
+ """
170
+ empty = {"pids": 0, "rss_bytes": None, "cpu_percent": None}
171
+ if root_pid is None:
172
+ return empty
173
+ children, stats = index if index is not None else build_index(proc_root)
174
+ if root_pid not in stats:
175
+ return empty
176
+ uptime = read_uptime(proc_root)
177
+ rss_pages = 0
178
+ cpu_ticks = 0
179
+ counted = 0
180
+ for pid in tree_pids(root_pid, children):
181
+ st = stats.get(pid)
182
+ if st is None:
183
+ continue
184
+ counted += 1
185
+ rss_pages += st["rss_pages"]
186
+ cpu_ticks += st["cpu_ticks"]
187
+ cpu_percent = None
188
+ if uptime is not None and _CLK_TCK:
189
+ wall = uptime - stats[root_pid]["starttime"] / _CLK_TCK
190
+ if wall > 0:
191
+ cpu_seconds = cpu_ticks / _CLK_TCK
192
+ cpu_percent = round(100.0 * cpu_seconds / wall, 1)
193
+ return {
194
+ "pids": counted,
195
+ "rss_bytes": rss_pages * _PAGE_SIZE,
196
+ "cpu_percent": cpu_percent,
197
+ }
198
+
199
+
200
+ def top_procs(
201
+ n: int = 5,
202
+ *,
203
+ proc_root: "str | os.PathLike[str]" = "/proc",
204
+ index: "tuple[dict[int, list[int]], dict[int, dict]] | None" = None,
205
+ ) -> "list[dict]":
206
+ """The n biggest processes on the HOST by RSS: [{pid, comm, rss_bytes}],
207
+ largest first.
208
+
209
+ Deliberately the whole `/proc` and not a pane tree: the memory tile's
210
+ question is container-wide, so when the charge really is resident the
211
+ operator needs to see whatever is holding it -- which is routinely not a
212
+ reviewer session at all (the daemon itself, a stray build, the shell that
213
+ started everything).
214
+
215
+ `index` is the same (children, stats) pair `tree_usage` takes, so a
216
+ dashboard build that already scanned `/proc` for its session table ranks
217
+ processes off that ONE snapshot instead of walking every PID twice.
218
+ Vanished PIDs never appear: `build_index` has already dropped them, and a
219
+ process that exits after the scan simply ranks with its last-known RSS.
220
+
221
+ Ties break on PID so the order is total -- an equal-RSS pair must not
222
+ reshuffle between polls and make a still list look like it is churning.
223
+ """
224
+ if n <= 0:
225
+ return []
226
+ _, stats = index if index is not None else build_index(proc_root)
227
+ ranked = sorted(stats.values(), key=lambda st: (-st["rss_pages"], st["pid"]))
228
+ return [
229
+ {
230
+ "pid": st["pid"],
231
+ "comm": st.get("comm") or "?",
232
+ "rss_bytes": st["rss_pages"] * _PAGE_SIZE,
233
+ }
234
+ for st in ranked[:n]
235
+ ]
236
+
237
+
238
+ def _read_int_file(path: Path) -> "int | None":
239
+ """One cgroup scalar file as an int, or None when it is absent, unreadable
240
+ or not a number. cgroup v2 writes the literal `max` in some files, which is
241
+ exactly the malformed case this returns None for."""
242
+ try:
243
+ return int(path.read_text().strip())
244
+ except (OSError, ValueError):
245
+ return None
246
+
247
+
248
+ def _read_stat_keys(path: Path, keys: "tuple[str, ...]") -> "dict[str, int | None]":
249
+ """The requested `key value` lines of a cgroup `*.stat` file. Every key is
250
+ present in the result whether or not the file had it, so the payload shape
251
+ never depends on the kernel version -- a missing or unparseable key is
252
+ None, the same thing the whole-file failure produces."""
253
+ out: "dict[str, int | None]" = {key: None for key in keys}
254
+ try:
255
+ text = path.read_text()
256
+ except (OSError, ValueError):
257
+ return out
258
+ wanted = set(keys)
259
+ for line in text.splitlines():
260
+ name, _, value = line.partition(" ")
261
+ if name not in wanted:
262
+ continue
263
+ try:
264
+ out[name] = int(value.strip())
265
+ except ValueError:
266
+ out[name] = None
267
+ return out
268
+
269
+
270
+ def _sum_or_none(*values: "int | None") -> "int | None":
271
+ """Sum only when EVERY part is known. A partial sum would be reported as a
272
+ whole bucket and read as a smaller cache than the container really holds --
273
+ an unknown number must stay unknown rather than become a wrong one."""
274
+ if any(value is None for value in values):
275
+ return None
276
+ return sum(value for value in values if value is not None)
277
+
278
+
279
+ def _drop_shmem(file_bytes: "int | None", shmem: "int | None") -> "int | None":
280
+ """The droppable part of `file`. `shmem` unknown means the droppable part
281
+ is unknown -- publishing `file` whole would be the reassuring-direction
282
+ error the split exists to prevent."""
283
+ if file_bytes is None or shmem is None:
284
+ return None
285
+ return max(0, file_bytes - shmem)
286
+
287
+
288
+ def cgroup_memory(
289
+ cgroup_root: "str | os.PathLike[str]" = "/sys/fs/cgroup",
290
+ ) -> dict:
291
+ """The container's own memory charge, split for the operator.
292
+
293
+ Answers the question a platform memory graph raises and cannot settle: a
294
+ flat multi-GB plateau is `charged` (cgroup v2 `memory.current`), but most
295
+ of it is routinely cache the kernel would drop under pressure, not a leak.
296
+ So we also report:
297
+
298
+ * `resident` -- `anon`, the process memory that is really in use.
299
+ * `reclaimable` -- `(file - shmem) + slab_reclaimable`: the page cache plus
300
+ the shrinkable kernel caches, the "not really in use" bucket.
301
+ * `shmem` -- broken out of `file` rather than left inside it, because it is
302
+ not droppable: tmpfs, shm segments and shared anonymous mmaps are
303
+ swap-backed, so with swap disabled or capped (the normal container case)
304
+ the kernel cannot reclaim them under pressure -- they are what the
305
+ container gets OOM-killed for. Counting them as reclaimable is the one
306
+ way this reader can be wrong in the REASSURING direction, telling an
307
+ operator "5.9 GB of cache, no leak" about pages nothing can free. Issue
308
+ #74's scope text prescribes `file + slab_reclaimable` verbatim; the same
309
+ sentence calls the result the "not really in use" bucket, and where the
310
+ formula and the words disagree the words are the requirement. Surfaced as
311
+ its own magnitude instead of folded into `resident`, because "5 GB of
312
+ tmpfs" and "5 GB of process heap" are different operator problems even
313
+ though neither is droppable.
314
+
315
+ The raw keys ride along for a spot check against an in-container
316
+ `memory.stat`. Note the summary numbers do NOT partition the charge --
317
+ kernel stacks, pagetables and sockets are charged too -- they are the
318
+ magnitudes an operator compares, not a balance sheet.
319
+
320
+ Never raises. On a host without cgroup v2 -- a dev laptop, macOS, a v1
321
+ cgroup tree -- every field is None and the console renders "unavailable";
322
+ the root is a parameter so all of that is testable off a fixture dir.
323
+ """
324
+ root = Path(cgroup_root)
325
+ charged = _read_int_file(root / _CGROUP_CURRENT)
326
+ stat = _read_stat_keys(root / _CGROUP_STAT, _CGROUP_STAT_KEYS)
327
+ out: "dict[str, int | None]" = {
328
+ "charged": charged,
329
+ "resident": stat["anon"],
330
+ # `shmem` is a SUBSET of `file` in cgroup v2, so it is subtracted, not
331
+ # added. Clamped at zero: the two keys are sampled from one read of one
332
+ # file and cannot legitimately cross, but a negative "cache" number
333
+ # would be a worse thing to render than a zero if a kernel ever
334
+ # disagreed with that.
335
+ "reclaimable": _sum_or_none(
336
+ _drop_shmem(stat["file"], stat["shmem"]), stat["slab_reclaimable"]
337
+ ),
338
+ }
339
+ out.update(stat)
340
+ return out
341
+
342
+
343
+ def disk_usage(path: "str | os.PathLike[str]") -> "dict | None":
344
+ """Workspace volume usage for the stat tile, or None if the path is
345
+ unreadable. Percent is used/total, rounded -- the meter the console fills."""
346
+ try:
347
+ usage = shutil.disk_usage(os.fspath(path))
348
+ except OSError:
349
+ return None
350
+ percent = round(100.0 * usage.used / usage.total, 1) if usage.total else None
351
+ return {
352
+ "total_bytes": usage.total,
353
+ "used_bytes": usage.used,
354
+ "free_bytes": usage.free,
355
+ "percent": percent,
356
+ }
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: alissa-tools-github-revloop
3
- Version: 0.16.14
3
+ Version: 0.16.16
4
4
  Summary: ALISSA-TOOLS-GITHUB-REVLOOP
5
5
  Home-page: https://alissa.app
6
6
  Author: Fahera
@@ -1,174 +0,0 @@
1
- """Per-session resource accounting off `/proc`, plus workspace disk usage.
2
-
3
- The sessions panel wants CPU% and RSS for each managed tmux session. tmux hands
4
- us a pane PID; the real work runs in that pane's child tree (a shell, the agent,
5
- its subprocesses), so we sum over the whole tree rooted at the pane PID.
6
-
7
- Two deliberate properties:
8
-
9
- * **Sample-free CPU.** Instantaneous CPU% needs two `/proc` reads spaced apart;
10
- a dashboard endpoint polled every ~10s should not block sampling. Instead we
11
- report the *cumulative average* since the pane started:
12
- `100 * (utime+stime)/CLK_TCK / (uptime - starttime/CLK_TCK)`. It answers "how
13
- hot has this session run over its life", which is the operator question, and
14
- needs a single read.
15
- * **Vanished-PID tolerant.** Between listing `/proc` and reading a PID's `stat`,
16
- the process can exit. Every read that fails (gone, permission, malformed) is
17
- skipped -- a disappearing reviewer contributes nothing and never raises.
18
-
19
- The `/proc` root is a parameter so the parsing is testable against a crafted
20
- fixture directory (including a PID that is indexed but whose `stat` was deleted
21
- mid-walk -- the vanished-PID case the mutation bar pins).
22
- """
23
-
24
- from __future__ import annotations
25
-
26
- import os
27
- import shutil
28
- from pathlib import Path
29
-
30
- _CLK_TCK = os.sysconf("SC_CLK_TCK") if hasattr(os, "sysconf") else 100
31
- _PAGE_SIZE = os.sysconf("SC_PAGE_SIZE") if hasattr(os, "sysconf") else 4096
32
-
33
- # Field offsets INTO the tail of /proc/<pid>/stat (everything after the
34
- # "(comm)" field, which itself is field 3 == index 0 of the tail). See
35
- # proc(5): utime=14, stime=15, starttime=22, rss=24 -> tail indices below.
36
- _TAIL_PPID = 1
37
- _TAIL_UTIME = 11
38
- _TAIL_STIME = 12
39
- _TAIL_STARTTIME = 19
40
- _TAIL_RSS = 21
41
- _TAIL_MIN_LEN = 22
42
-
43
-
44
- def read_stat(proc_root: "str | os.PathLike[str]", pid: int) -> "dict | None":
45
- """Parse one `/proc/<pid>/stat`, or None if the PID has vanished or the
46
- line is malformed. The `comm` field can contain spaces and parentheses, so
47
- we split on the LAST ')' before tokenising -- the canonical safe parse."""
48
- try:
49
- data = (Path(proc_root) / str(pid) / "stat").read_text()
50
- except (OSError, ValueError):
51
- return None
52
- rparen = data.rfind(")")
53
- if rparen == -1:
54
- return None
55
- tail = data[rparen + 1:].split()
56
- if len(tail) < _TAIL_MIN_LEN:
57
- return None
58
- try:
59
- return {
60
- "pid": int(pid),
61
- "ppid": int(tail[_TAIL_PPID]),
62
- "cpu_ticks": int(tail[_TAIL_UTIME]) + int(tail[_TAIL_STIME]),
63
- "starttime": int(tail[_TAIL_STARTTIME]),
64
- "rss_pages": int(tail[_TAIL_RSS]),
65
- }
66
- except (ValueError, IndexError):
67
- return None
68
-
69
-
70
- def read_uptime(proc_root: "str | os.PathLike[str]") -> "float | None":
71
- try:
72
- return float((Path(proc_root) / "uptime").read_text().split()[0])
73
- except (OSError, ValueError, IndexError):
74
- return None
75
-
76
-
77
- def build_index(
78
- proc_root: "str | os.PathLike[str]",
79
- ) -> "tuple[dict[int, list[int]], dict[int, dict]]":
80
- """Scan `/proc` once: return (ppid -> [child pids], pid -> stat). Numeric
81
- entries only; anything that vanishes mid-scan is silently skipped."""
82
- children: dict[int, list[int]] = {}
83
- stats: dict[int, dict] = {}
84
- try:
85
- entries = os.listdir(proc_root)
86
- except OSError:
87
- return children, stats
88
- for name in entries:
89
- if not name.isdigit():
90
- continue
91
- st = read_stat(proc_root, int(name))
92
- if st is None:
93
- continue
94
- stats[st["pid"]] = st
95
- children.setdefault(st["ppid"], []).append(st["pid"])
96
- return children, stats
97
-
98
-
99
- def tree_pids(root_pid: int, children: "dict[int, list[int]]") -> "set[int]":
100
- """Every PID in the subtree rooted at `root_pid` (inclusive). Iterative and
101
- cycle-guarded -- a corrupt ppid loop can never spin forever."""
102
- seen: set[int] = set()
103
- stack = [root_pid]
104
- while stack:
105
- pid = stack.pop()
106
- if pid in seen:
107
- continue
108
- seen.add(pid)
109
- stack.extend(children.get(pid, ()))
110
- return seen
111
-
112
-
113
- def tree_usage(
114
- root_pid: "int | None",
115
- *,
116
- proc_root: "str | os.PathLike[str]" = "/proc",
117
- index: "tuple[dict[int, list[int]], dict[int, dict]] | None" = None,
118
- ) -> dict:
119
- """CPU% (cumulative average since the pane started) and summed RSS for the
120
- whole process tree under `root_pid`. Returns None/None when the root has
121
- already vanished -- "unknown", the same thing the caller shows for a
122
- session with no pane at all, so the column reads consistently.
123
-
124
- `index` is the (children, stats) pair from `build_index`. Pass it when
125
- accounting SEVERAL trees from one snapshot of `/proc`: the index is
126
- identical for every session in one dashboard build, and rebuilding it per
127
- session makes the walk O(sessions x processes) -- on a busy reviewer host
128
- thousands of `stat` reads every poll, stolen from the reviewer agents the
129
- console exists to watch. Omitted, it is built once for this call.
130
- """
131
- empty = {"pids": 0, "rss_bytes": None, "cpu_percent": None}
132
- if root_pid is None:
133
- return empty
134
- children, stats = index if index is not None else build_index(proc_root)
135
- if root_pid not in stats:
136
- return empty
137
- uptime = read_uptime(proc_root)
138
- rss_pages = 0
139
- cpu_ticks = 0
140
- counted = 0
141
- for pid in tree_pids(root_pid, children):
142
- st = stats.get(pid)
143
- if st is None:
144
- continue
145
- counted += 1
146
- rss_pages += st["rss_pages"]
147
- cpu_ticks += st["cpu_ticks"]
148
- cpu_percent = None
149
- if uptime is not None and _CLK_TCK:
150
- wall = uptime - stats[root_pid]["starttime"] / _CLK_TCK
151
- if wall > 0:
152
- cpu_seconds = cpu_ticks / _CLK_TCK
153
- cpu_percent = round(100.0 * cpu_seconds / wall, 1)
154
- return {
155
- "pids": counted,
156
- "rss_bytes": rss_pages * _PAGE_SIZE,
157
- "cpu_percent": cpu_percent,
158
- }
159
-
160
-
161
- def disk_usage(path: "str | os.PathLike[str]") -> "dict | None":
162
- """Workspace volume usage for the stat tile, or None if the path is
163
- unreadable. Percent is used/total, rounded -- the meter the console fills."""
164
- try:
165
- usage = shutil.disk_usage(os.fspath(path))
166
- except OSError:
167
- return None
168
- percent = round(100.0 * usage.used / usage.total, 1) if usage.total else None
169
- return {
170
- "total_bytes": usage.total,
171
- "used_bytes": usage.used,
172
- "free_bytes": usage.free,
173
- "percent": percent,
174
- }