workmap 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
workmap/procs.py ADDED
@@ -0,0 +1,504 @@
1
+ """Process and memory facts, straight from the OS.
2
+
3
+ Portable POSIX territory: ps, lsof, vm_stat, signals. Nothing here knows what
4
+ a Terminal window is."""
5
+ from __future__ import annotations
6
+
7
+ import os
8
+ import re
9
+ import signal
10
+ import subprocess
11
+ import time
12
+ from collections import defaultdict
13
+ from pathlib import Path
14
+
15
+
16
+ def run(cmd: list[str], *, timeout: float = 8.0, **kwargs) -> str:
17
+ """Capture a command's stdout. Always bounded, lsof on a stale network
18
+ mount, or a wedged Finder, would otherwise hang the whole tool forever."""
19
+ return subprocess.check_output(
20
+ cmd, text=True, stderr=subprocess.DEVNULL, timeout=timeout, **kwargs
21
+ )
22
+
23
+
24
+ def run_safe(cmd: list[str], *, timeout: float = 8.0, default: str = "") -> str:
25
+ """run() that degrades to `default` instead of propagating."""
26
+ try:
27
+ return run(cmd, timeout=timeout)
28
+ except (subprocess.CalledProcessError, subprocess.TimeoutExpired,
29
+ OSError, ValueError):
30
+ return default
31
+
32
+
33
+ def run_partial(cmd: list[str], *, timeout: float = 8.0) -> str:
34
+ """Capture stdout and keep it even when the command exits non-zero.
35
+
36
+ For tools that report "I couldn't do all of it" through the exit status
37
+ while still printing everything they *could* do. lsof is the reason this
38
+ exists: asked about a batch of pids it exits 1 if a single one is
39
+ unreadable. `login`, which sits on every macOS Terminal tab, is
40
+ root-owned and always unreadable. Treating that as total failure threw
41
+ away the working directory of every other process in the batch, which is
42
+ how every project ended up in the "other" bucket.
43
+ """
44
+ try:
45
+ proc = subprocess.run(
46
+ cmd, capture_output=True, text=True, timeout=timeout, check=False
47
+ )
48
+ except (subprocess.TimeoutExpired, OSError, ValueError):
49
+ return ""
50
+ return proc.stdout or ""
51
+
52
+
53
+ # `lstart` is five whitespace-separated tokens on macOS, always, even for a
54
+ # single-digit day, where ps pads with a second space that split() collapses.
55
+ # That fixed width is what lets it sit in front of the command in one pass.
56
+ _LSTART_FIELDS = 5
57
+ _PS_FORMAT = "pid=,ppid=,rss=,tty=,lstart=,command="
58
+
59
+
60
+ def _start_time(raw: str) -> str:
61
+ """A start time in one spelling, whoever read it.
62
+
63
+ ps pads the day to two columns, so a process started on the 8th reports
64
+ "Wed Jul 8 07:14:41 2026" with two spaces. The scan reads lstart with
65
+ split(), which collapses that; the kill-time recheck read it with
66
+ split(None, 1), which kept it. The two never matched again, and since a
67
+ start time is what identity means here, every process started on the first
68
+ nine days of a month looked like a recycled pid and was spared. Measured on
69
+ this machine: 200 of 400 live pids. `workmap kill` reported the memory it
70
+ had freed and had signalled nothing.
71
+
72
+ One function, both readers, so they cannot drift apart again.
73
+ """
74
+ return " ".join(raw.split())
75
+
76
+
77
+ def process_snapshot() -> tuple[list[dict], dict[int, list[int]]]:
78
+ """One `ps` pass for both the process table and the parent→child map.
79
+
80
+ These used to be two separate full-table scans of the same data.
81
+
82
+ An empty table means `ps` did not answer, because a machine always has
83
+ processes on it. Callers have to treat it as a failure rather than as an
84
+ idle machine: everything the scan knows is derived from this, so no rows
85
+ means no windows, no owners and no orphans, which draws exactly the desk
86
+ of somebody with nothing open. `ps` missing its timeout on a loaded
87
+ machine is the likely way in, and the Terminal query is fine at the time,
88
+ so the guard that already exists for that does not fire.
89
+ """
90
+ out = run_safe(["ps", "-axo", _PS_FORMAT], timeout=6.0)
91
+ rows: list[dict] = []
92
+ kids: dict[int, list[int]] = defaultdict(list)
93
+ fields = 5 + _LSTART_FIELDS
94
+ for line in out.splitlines():
95
+ line = line.strip()
96
+ if not line:
97
+ continue
98
+ parts = line.split(None, fields - 1)
99
+ if len(parts) < fields:
100
+ continue
101
+ pid_s, ppid_s, rss_s, tty = parts[:4]
102
+ started = _start_time(" ".join(parts[4:4 + _LSTART_FIELDS]))
103
+ cmd = parts[4 + _LSTART_FIELDS]
104
+ try:
105
+ pid, ppid, rss = int(pid_s), int(ppid_s), int(rss_s)
106
+ except ValueError:
107
+ continue
108
+ rows.append({
109
+ "pid": pid,
110
+ "ppid": ppid,
111
+ "rss_kb": rss,
112
+ "tty": tty if tty != "??" else "",
113
+ "started": started,
114
+ "cmd": cmd,
115
+ })
116
+ kids[ppid].append(pid)
117
+ return rows, kids
118
+
119
+
120
+ def process_table() -> list[dict]:
121
+ return process_snapshot()[0]
122
+
123
+
124
+ def children_map() -> dict[int, list[int]]:
125
+ return process_snapshot()[1]
126
+
127
+
128
+ def descendant_pids(root: int, kids: dict[int, list[int]]) -> set[int]:
129
+ seen: set[int] = set()
130
+ stack = [root]
131
+ while stack:
132
+ pid = stack.pop()
133
+ if pid in seen:
134
+ continue
135
+ seen.add(pid)
136
+ stack.extend(kids.get(pid, []))
137
+ return seen
138
+
139
+
140
+ def _parse_size_kb(text: str) -> int | None:
141
+ """top's size column: "26M", "1261M", "6167M+", "512K", "1.2G"."""
142
+ s = text.strip().rstrip("+-")
143
+ if not s:
144
+ return None
145
+ unit, digits = s[-1].upper(), s[:-1]
146
+ scale = {"K": 1, "M": 1024, "G": 1024 * 1024, "B": 1 / 1024}.get(unit)
147
+ if scale is None:
148
+ unit, digits, scale = "B", s, 1 / 1024
149
+ try:
150
+ return int(float(digits) * scale)
151
+ except ValueError:
152
+ return None
153
+
154
+
155
+ def footprint_by_pid() -> dict[int, int]:
156
+ """Per-pid phys_footprint in KB, the number Activity Monitor shows.
157
+
158
+ `ps -o rss` is not what it sounds like on macOS. It reports resident
159
+ *dirty* pages: it excludes the dyld shared cache a process has mapped
160
+ (~180 MB of shared __TEXT per process, which is right: that memory is
161
+ shared, not per-process), but it also excludes anything the compressor has
162
+ taken, which is wrong. On a machine under pressure that is most of the
163
+ heap: a node process here showed 34 MB resident against 363 MB compressed.
164
+
165
+ phys_footprint counts compressed pages and still excludes shared clean
166
+ ones, so it is both the honest number and the comparable one. One `top`
167
+ pass costs ~0.9s for the whole table; callers overlap it with other work.
168
+ Returns {} if top is unavailable, callers fall back to ps.
169
+ """
170
+ out: dict[int, int] = {}
171
+ raw = run_partial(["top", "-l", "1", "-stats", "pid,mem"], timeout=15.0)
172
+ started = False
173
+ for line in raw.splitlines():
174
+ fields = line.split()
175
+ if not started:
176
+ # Skip the system summary; the table starts at its header row.
177
+ started = bool(fields) and fields[0] == "PID"
178
+ continue
179
+ if len(fields) < 2 or not fields[0].isdigit():
180
+ continue
181
+ kb = _parse_size_kb(fields[1])
182
+ if kb is not None:
183
+ out[int(fields[0])] = kb
184
+ return out
185
+
186
+
187
+ def tree_mb(pids: set[int], rows: list[dict],
188
+ footprint: dict[int, int] | None = None) -> int:
189
+ """Memory held by a process tree, in MB.
190
+
191
+ Summing across a tree is only honest because the per-process figure
192
+ excludes shared pages: the framework a parent and six children all map is
193
+ counted once by the kernel, not seven times. See footprint_by_pid().
194
+ """
195
+ by_pid = {r["pid"]: r["rss_kb"] for r in rows}
196
+ fp = footprint or {}
197
+ # Per-pid fallback, not all-or-nothing: top and ps are separate samples, so
198
+ # a process can appear in one and not the other.
199
+ return sum(fp.get(pid) or by_pid.get(pid, 0) for pid in pids) // 1024
200
+
201
+
202
+ def cwd_for_pid(pid: int) -> Path | None:
203
+ return cwd_for_pids([pid]).get(pid)
204
+
205
+
206
+ def cwd_for_pids(pids: list[int]) -> dict[int, Path]:
207
+ """One lsof for many pids (per-pid lsof was a big slice of refresh cost)."""
208
+ out: dict[int, Path] = {}
209
+ uniq = []
210
+ seen: set[int] = set()
211
+ for pid in pids:
212
+ if pid and pid not in seen:
213
+ seen.add(pid)
214
+ uniq.append(pid)
215
+ if not uniq:
216
+ return out
217
+ # lsof has an argument length limit, so chunk
218
+ for i in range(0, len(uniq), 40):
219
+ chunk = uniq[i : i + 40]
220
+ # lsof exits non-zero when any pid in the batch is gone or unreadable,
221
+ # having already printed the rest. Keep stdout regardless. It can
222
+ # also block on stale network mounts, hence the timeout.
223
+ raw = run_partial(
224
+ ["lsof", "-a", "-d", "cwd", "-p", ",".join(str(p) for p in chunk), "-Fn"],
225
+ timeout=6.0,
226
+ )
227
+ if not raw:
228
+ continue
229
+ cur_pid: int | None = None
230
+ for line in raw.splitlines():
231
+ if line.startswith("p"):
232
+ try:
233
+ cur_pid = int(line[1:])
234
+ except ValueError:
235
+ cur_pid = None
236
+ elif line.startswith("n") and cur_pid is not None:
237
+ p = Path(line[1:])
238
+ if cur_pid not in out and _still_there(p):
239
+ out[cur_pid] = p
240
+ return out
241
+
242
+
243
+ def _still_there(path: Path) -> bool:
244
+ """Does this directory exist? Only "no such file" counts as no.
245
+
246
+ The check is here to drop a working directory that has been deleted since
247
+ the process entered it. A directory nobody may traverse is a different
248
+ answer: lsof has just said a live process is sitting in it, so it is
249
+ there.
250
+
251
+ Asked with stat() rather than exists(), because exists() does not mean the
252
+ same thing on the two interpreters this ships against. On 3.9, the floor
253
+ and the one it installs into, it propagates EACCES, so one project
254
+ directory with its permissions changed took down every scan, which is
255
+ every keypress in the desk. On 3.12 and later it swallows that and answers
256
+ False, which is quieter and still wrong: the process is dropped and its
257
+ project goes unnamed. stat() raises both, and the two are told apart here,
258
+ so the answer is the same wherever it runs.
259
+ """
260
+ try:
261
+ path.stat()
262
+ return True
263
+ except FileNotFoundError:
264
+ return False
265
+ except OSError:
266
+ return True
267
+
268
+
269
+ def unix_socket_paths(pids: list[int]) -> dict[int, list[str]]:
270
+ """Which unix-domain sockets each of these processes is listening on.
271
+
272
+ cwd_for_pids() batched the same way for the same reason: one lsof, not one
273
+ per pid. This one answers "where do I talk to that server", which is how
274
+ panes.py reaches a multiplexer without guessing at a socket directory.
275
+
276
+ Only names that are paths are kept. lsof reports the other end of a
277
+ connected socket as a kernel address (`->0xcfedb109393bb6a6`), which names
278
+ nothing anything can be asked through.
279
+ """
280
+ out: dict[int, list[str]] = {}
281
+ uniq = [p for p in dict.fromkeys(pids) if p]
282
+ if not uniq:
283
+ return out
284
+ for i in range(0, len(uniq), 40):
285
+ chunk = uniq[i:i + 40]
286
+ raw = run_partial(
287
+ ["lsof", "-a", "-U", "-p", ",".join(str(p) for p in chunk), "-Fn"],
288
+ timeout=6.0,
289
+ )
290
+ cur_pid: int | None = None
291
+ for line in raw.splitlines():
292
+ if line.startswith("p"):
293
+ try:
294
+ cur_pid = int(line[1:])
295
+ except ValueError:
296
+ cur_pid = None
297
+ elif line.startswith("n") and cur_pid is not None:
298
+ name = line[1:]
299
+ # One socket can appear on several descriptors, so the same
300
+ # path comes back more than once and would be asked twice.
301
+ if name.startswith("/") and name not in out.get(cur_pid, ()):
302
+ out.setdefault(cur_pid, []).append(name)
303
+ return out
304
+
305
+
306
+ def mem_summary() -> dict:
307
+ free_mb: int | str = "?"
308
+ swap_mb: str = "?"
309
+ try:
310
+ vm = run(["vm_stat"])
311
+ page_size = 4096
312
+ m = re.search(r"page size of (\d+) bytes", vm)
313
+ if m:
314
+ page_size = int(m.group(1))
315
+ m = re.search(r"Pages free:\s+(\d+)", vm)
316
+ if m:
317
+ free_mb = int(m.group(1)) * page_size // 1024 // 1024
318
+ except Exception:
319
+ pass
320
+ try:
321
+ swap = run(["sysctl", "-n", "vm.swapusage"])
322
+ m = re.search(r"used = ([0-9.]+)M", swap)
323
+ if m:
324
+ swap_mb = m.group(1)
325
+ except Exception:
326
+ pass
327
+ return {"free_mb": free_mb, "swap_mb": swap_mb}
328
+
329
+
330
+ def pid_alive(pid: int) -> bool:
331
+ try:
332
+ os.kill(pid, 0)
333
+ return True
334
+ except ProcessLookupError:
335
+ return False
336
+ except PermissionError:
337
+ return True # exists, just not ours
338
+
339
+
340
+ def killable_pids(pids: list[int]) -> list[int]:
341
+ """Drop pids we must never signal: ourselves, our parent, our process
342
+ group leader, and init.
343
+
344
+ Not "our own ancestry", which is what this used to claim: it has no
345
+ process table, so it cannot walk one. Everything further up is protected
346
+ a layer earlier, by scan.build_projects(), which never offers a process
347
+ descended from the terminal or from a running application. This is the
348
+ last net, and it is worth saying what it actually catches.
349
+ """
350
+ forbidden = {0, 1, os.getpid(), os.getppid()}
351
+ try:
352
+ forbidden.add(os.getpgid(0))
353
+ except OSError:
354
+ pass
355
+ out: list[int] = []
356
+ for pid in pids:
357
+ if pid and pid > 1 and pid not in forbidden and pid not in out:
358
+ out.append(pid)
359
+ return out
360
+
361
+
362
+ def current_starts(pids: list[int]) -> dict[int, str]:
363
+ """When each of these pids started, for pids we can read.
364
+
365
+ `lstart` goes last here so its spaces cannot be confused with a field
366
+ boundary.
367
+ """
368
+ out: dict[int, str] = {}
369
+ if not pids:
370
+ return out
371
+ raw = run_partial(
372
+ ["ps", "-o", "pid=,lstart=", "-p", ",".join(str(p) for p in pids)],
373
+ timeout=6.0,
374
+ )
375
+ for line in raw.splitlines():
376
+ parts = line.strip().split(None, 1)
377
+ if len(parts) == 2 and parts[0].isdigit():
378
+ out[int(parts[0])] = _start_time(parts[1])
379
+ return out
380
+
381
+
382
+ def still_the_same(pids: list[int], expected: dict[int, str],
383
+ live: dict[int, str] | None = None) -> list[int]:
384
+ """Drop pids that are not the process we recorded.
385
+
386
+ A scan can sit on screen for hours before the user acts on it, and macOS
387
+ hands pids back out after ~99k more processes. Without this, `k` signals
388
+ whatever inherited the number, which on a dev machine is something that
389
+ started *since* the scan: the newest thing the user is working on.
390
+
391
+ Identity is the start time, not the command line. A command line is not an
392
+ identity: a process can rewrite its own argv (plenty of servers do), which
393
+ would make a live process look recycled and quietly un-killable. A start
394
+ time cannot be rewritten, and a recycled pid necessarily started later
395
+ than the scan that recorded it.
396
+
397
+ A pid `ps` answered about but did not list has exited, and keeping it costs
398
+ nothing: os.kill reports it already gone. A `ps` that answered about *none*
399
+ of them is a different thing, and used to be treated as the same one. It is
400
+ not evidence that all of them are fine, it is the absence of evidence, and
401
+ this check is the only thing standing between an hours-old scan and a
402
+ signal sent to whatever now holds those numbers. So nothing is signalled:
403
+ if they really are all gone there is nothing to lose, and if `ps` merely
404
+ timed out there is everything.
405
+ """
406
+ if not expected:
407
+ return pids
408
+ if live is None:
409
+ live = current_starts(pids)
410
+ if pids and not live:
411
+ return []
412
+ out = []
413
+ for pid in pids:
414
+ want = expected.get(pid)
415
+ now = live.get(pid)
416
+ if want is None:
417
+ # No identity was recorded for this pid, so there is nothing to
418
+ # compare against and this guard cannot vouch for it. It used to
419
+ # be kept, which let exactly the pid the guard exists for through
420
+ # the one check meant to catch it. The scanner only records a
421
+ # start time for pids `ps` listed, so a pid it could not read is
422
+ # spared instead. Same reasoning as the empty-`live` case above:
423
+ # if it really has gone there is nothing to lose by not
424
+ # signalling it, and if it has been reused there is everything.
425
+ continue
426
+ if now is None or now == want:
427
+ out.append(pid)
428
+ return out
429
+
430
+
431
+ def kill_pids(pids: list[int], expected: dict[int, str] | None = None, *,
432
+ dry_run: bool = False, outcomes: list[dict] | None = None) -> int:
433
+ """SIGTERM, wait, then SIGKILL only what is still alive.
434
+
435
+ The liveness re-check matters: the old code SIGKILLed every pid
436
+ unconditionally after the grace period, so a pid that had already exited
437
+ and been recycled by the OS could take an unrelated process with it. The
438
+ same hazard applies to the opening SIGTERM, which is what `expected`
439
+ guards. See still_the_same().
440
+
441
+ `dry_run` runs every check and reports what it would do without sending a
442
+ signal. `outcomes`, if given, is filled with one record per pid so the
443
+ caller can write them to the audit log; the reason a pid was spared is
444
+ part of the record, since "nothing happened" has several causes worth
445
+ telling apart.
446
+ """
447
+ log = outcomes if outcomes is not None else []
448
+ asked = list(dict.fromkeys(p for p in pids if p))
449
+ # Read the identities once, so the log can tell "this pid is not the
450
+ # process we recorded" apart from "we could not find out". Both spare the
451
+ # pid; only one of them is a fact about the pid.
452
+ live = current_starts(asked) if (asked and expected) else {}
453
+ unchecked = bool(asked) and bool(expected) and not live
454
+ survivors = still_the_same(asked, expected or {}, live)
455
+ uniq = killable_pids(survivors)
456
+ for pid in asked:
457
+ if pid not in survivors:
458
+ if unchecked:
459
+ why = "skipped-unchecked"
460
+ elif expected and pid not in expected:
461
+ # Spared because nothing was recorded about it, which is a
462
+ # different fact from "the pid was reused" and used to be
463
+ # reported as one.
464
+ why = "skipped-unrecorded"
465
+ else:
466
+ why = "skipped-recycled"
467
+ log.append({"pid": pid, "signal": None, "result": why})
468
+ elif pid not in uniq:
469
+ log.append({"pid": pid, "signal": None, "result": "skipped-protected"})
470
+
471
+ if dry_run:
472
+ for pid in uniq:
473
+ log.append({"pid": pid, "signal": "SIGTERM", "result": "dry-run"})
474
+ return len(uniq)
475
+
476
+ n = 0
477
+ for pid in uniq:
478
+ try:
479
+ os.kill(pid, signal.SIGTERM)
480
+ n += 1
481
+ log.append({"pid": pid, "signal": "SIGTERM", "result": "sent"})
482
+ except ProcessLookupError:
483
+ log.append({"pid": pid, "signal": "SIGTERM", "result": "already-gone"})
484
+ except PermissionError:
485
+ log.append({"pid": pid, "signal": "SIGTERM", "result": "not-permitted"})
486
+ if not uniq:
487
+ return 0
488
+ # Give them a moment to go quietly, then re-check before escalating.
489
+ deadline = time.time() + 0.5
490
+ while time.time() < deadline:
491
+ if not any(pid_alive(p) for p in uniq):
492
+ return n
493
+ time.sleep(0.05)
494
+ for pid in uniq:
495
+ if not pid_alive(pid):
496
+ continue
497
+ try:
498
+ os.kill(pid, signal.SIGKILL)
499
+ log.append({"pid": pid, "signal": "SIGKILL", "result": "sent"})
500
+ except ProcessLookupError:
501
+ log.append({"pid": pid, "signal": "SIGKILL", "result": "already-gone"})
502
+ except PermissionError:
503
+ log.append({"pid": pid, "signal": "SIGKILL", "result": "not-permitted"})
504
+ return n