workmap 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
workmap/scan.py ADDED
@@ -0,0 +1,413 @@
1
+ """Build the desk map: which projects exist, and what each one is holding.
2
+
3
+ This is the one expensive call in the tool (~1s, dominated by AppleScript).
4
+ Everything downstream takes the result as an argument rather than calling it
5
+ again. See actions.py."""
6
+ from __future__ import annotations
7
+
8
+ import re
9
+ import threading
10
+ from collections import defaultdict
11
+ from pathlib import Path
12
+
13
+ from . import multiplexer, procs, terminal
14
+ from .config import (ORPHAN_FLOOR_MB, build_project_index, load_config,
15
+ profile_for, same_name)
16
+ from .model import (REPORTABLE_HOLDS, SHELLS, Hold, Project, Session,
17
+ _program_name, classify_cmd, clean_label, held_note,
18
+ holder, is_gui_app, is_self_process, multiplexer_name,
19
+ strongest, traceable, tty_short)
20
+ from .procs import (cwd_for_pid, cwd_for_pids, descendant_pids,
21
+ footprint_by_pid, process_snapshot, tree_mb)
22
+
23
+
24
+ def title_label(raw_title: str, procs_on_tty: list[dict]) -> str:
25
+ """The human-meaningful part of a Terminal window name.
26
+
27
+ Terminal builds the name from em-dash separated pieces: the working
28
+ directory or user, any custom title (which is where an agent writes what
29
+ it is working on), and the commands currently running on the tty. Only the
30
+ middle piece is worth showing, so the other two have to be recognised.
31
+
32
+ The running-command piece used to be matched against a hard-coded list of
33
+ tool names, which failed the moment anything else was in the foreground.
34
+ A window doing chess work showed up as "sourcekit-lsp ◂ claude
35
+ --dangerously-skip-permissions". We already know what is running on this
36
+ tty, so ask the process table instead of guessing.
37
+ """
38
+ title = re.sub(r"\s—\s*\d+×\d+\s*$", "", raw_title or "")
39
+ if not title.strip():
40
+ return ""
41
+ running = {_program_name(r["cmd"].split()[0]) for r in procs_on_tty
42
+ if r["cmd"].split()}
43
+ running.discard("")
44
+
45
+ def is_command_list(part: str) -> bool:
46
+ """Does this piece just name processes that are running right here?"""
47
+ # Terminal joins concurrent commands with "◂".
48
+ names = [chunk.strip().split()[0] for chunk in part.split("◂")
49
+ if chunk.strip()]
50
+ return bool(names) and all(_program_name(n) in running for n in names)
51
+
52
+ parts = [p.strip() for p in title.split("—") if p.strip()]
53
+ kept = [p for p in parts if not is_command_list(p)]
54
+ if not kept:
55
+ return ""
56
+ # Terminal only prepends the directory/user piece when it has something to
57
+ # prepend it to. So a lone survivor that was the *first* of several is that
58
+ # prefix and says nothing. A title with no separators at all is a
59
+ # custom one somebody set, and is the whole point.
60
+ if len(parts) > 1 and kept == parts[:1]:
61
+ return ""
62
+ pretty = kept[-1]
63
+ if pretty in ("zsh", "-zsh", "login", Path.home().name):
64
+ return ""
65
+ pretty = clean_label(pretty)
66
+ # A title we set ourselves reads "project · what"; keep the what.
67
+ if " · " in pretty:
68
+ pretty = pretty.split(" · ", 1)[1].strip() or pretty
69
+ return pretty
70
+
71
+
72
+ def build_projects(*, held: list[Hold] | None = None) -> list[Project]:
73
+ """The desk: every project with something running, and what it is holding.
74
+
75
+ `held` is filled, if given, with one Hold per process this scan recognised
76
+ and declined to offer because something has a claim on it that the desk
77
+ does not show. That is how a caller can tell "nothing is running here"
78
+ apart from "something is running here and a tmux pane has it", which read
79
+ identically before and are not the same news. Same out-parameter shape as
80
+ procs.kill_pids(outcomes=), for the same reason: the answer is a side
81
+ channel, not the thing the caller asked for.
82
+ """
83
+ cfg = load_config()
84
+ # One directory listing per configured root, and one resolve() per project
85
+ # in them. Measured at 2ms for 44 projects against a scan of about a
86
+ # second, which is what it costs to be able to answer for a project
87
+ # reached through a symlink. See config.ProjectIndex.
88
+ index = build_project_index()
89
+ # `top` and AppleScript each cost about a second and neither needs the
90
+ # other, so the memory sample runs while Terminal is being queried.
91
+ mem_sample: dict[int, int] = {}
92
+
93
+ def _sample_memory() -> None:
94
+ mem_sample.update(footprint_by_pid())
95
+
96
+ sampler = threading.Thread(target=_sample_memory, daemon=True)
97
+ sampler.start()
98
+ tabs = terminal.list_terminal_tabs()
99
+ rows, kids = process_snapshot() # one ps pass, not two
100
+ if not rows:
101
+ # A machine always has processes on it, so no rows means `ps` did not
102
+ # answer. Everything below is derived from it, so carrying on draws
103
+ # the desk of somebody with nothing open and offers nothing to quit:
104
+ # a silent all clear on a machine nobody managed to read. Same shape
105
+ # as the Terminal guard further down, and it has to be separate,
106
+ # because Terminal answering fine is exactly the case.
107
+ return []
108
+ by_tty: dict[str, list[dict]] = defaultdict(list)
109
+ for r in rows:
110
+ if r["tty"]:
111
+ by_tty[r["tty"]].append(r)
112
+
113
+ projects: dict[str, Project] = {}
114
+ claimed_pids: set[int] = set()
115
+
116
+ def ensure(name: str, path: Path | None) -> Project:
117
+ if name not in projects:
118
+ projects[name] = Project(
119
+ name=name,
120
+ path=path,
121
+ # profile_for, not ensure_: a scan must not write to disk.
122
+ profile=profile_for(name, cfg),
123
+ )
124
+ elif path and projects[name].path is None:
125
+ projects[name].path = path
126
+ return projects[name]
127
+
128
+ # Batch cwd lookups for every tty process once (not lsof-per-pid in the loop).
129
+ tty_pids: list[int] = []
130
+ for tab in tabs:
131
+ for r in by_tty.get(tty_short(tab["tty"]), []):
132
+ tty_pids.append(r["pid"])
133
+ cwd_by_pid = cwd_for_pids(tty_pids)
134
+
135
+ # Wait for the memory sample as late as possible: `top` costs about 0.9s
136
+ # and is now the longest single thing in a scan, so every call above that
137
+ # does not need it belongs in front of this line rather than behind it.
138
+ sampler.join(timeout=15.0)
139
+
140
+ for tab in tabs:
141
+ tty = tty_short(tab["tty"])
142
+ procs_on_tty = by_tty.get(tty, [])
143
+ cwd = None
144
+ # The shell first, because its working directory is the one the person
145
+ # is in. Read from the program name rather than by looking for "zsh"
146
+ # anywhere in the line: a tmux server started with a `-c <dir>` in a
147
+ # project whose path contains it matched, and so did every command
148
+ # with a zsh script somewhere in its arguments.
149
+ for r in sorted(procs_on_tty,
150
+ key=lambda x: 0 if _program_name(
151
+ (x["cmd"].split() or [""])[0]) in SHELLS else 1):
152
+ cwd = cwd_by_pid.get(r["pid"])
153
+ if cwd:
154
+ break
155
+ # A window belongs on the desk wherever it is, so a directory that is
156
+ # not a project still gets somewhere to go. An orphan gets no such
157
+ # fallback: no project means not offered, which is the safe direction.
158
+ found = index.of_path(cwd)
159
+ name, path = ((found.name, found.path) if found
160
+ else index.bucket_for(cwd))
161
+ if cwd is None and tab["title"]:
162
+ # Terminal puts the working directory in front of its own title,
163
+ # so a tab whose processes will not say where they are can still
164
+ # be placed by what it is called. Through the index like every
165
+ # other reading of a name, so it can only name a project that is
166
+ # there: it used to accept any directory at all under a root.
167
+ prefix = tab["title"].split("—", 1)[0].strip()
168
+ named = index.named(prefix) if prefix else None
169
+ if named is not None:
170
+ name, path = named.name, named.path
171
+
172
+ label = "shell"
173
+ session_pids: set[int] = set()
174
+ for r in procs_on_tty:
175
+ session_pids |= descendant_pids(r["pid"], kids)
176
+ kind = classify_cmd(r["cmd"])
177
+ if kind in ("claude", "codex", "cursor-agent"):
178
+ label = kind
179
+ elif kind and label == "shell":
180
+ label = kind
181
+ if not session_pids:
182
+ session_pids = {r["pid"] for r in procs_on_tty}
183
+
184
+ # Skip the workmap desk window itself: the one running workmap, not
185
+ # every window that mentions it. This asks whether a process here *is*
186
+ # workmap, because the cost of being wrong is a live window vanishing
187
+ # off the desk with its memory uncounted. The looser _is_self(), which
188
+ # reads every token, is for the other question: never make this a
189
+ # target. Two decisions, two costs, two rules.
190
+ if any(is_self_process(r["cmd"]) for r in procs_on_tty):
191
+ claimed_pids |= session_pids
192
+ continue
193
+
194
+ pretty = title_label(tab["title"], procs_on_tty) or label
195
+
196
+ mb = tree_mb(session_pids, rows, mem_sample)
197
+ claimed_pids |= session_pids
198
+ proj = ensure(name, path)
199
+ proj.sessions.append(Session(
200
+ kind="window",
201
+ label=pretty if pretty != label else label,
202
+ mb=mb,
203
+ window_id=tab["window_id"],
204
+ tty=tty,
205
+ title=tab["title"],
206
+ profile=tab.get("profile") or None,
207
+ pids=sorted(session_pids),
208
+ cmds={r["pid"]: r["cmd"] for r in rows if r["pid"] in session_pids},
209
+ starts={r["pid"]: r["started"] for r in rows
210
+ if r["pid"] in session_pids},
211
+ ))
212
+
213
+ # "Nothing owns this" is only knowable if we could ask what the windows
214
+ # own. When the Terminal query fails every process on the machine looks
215
+ # unowned, which is the one situation where a kill list must not be built:
216
+ # the tool would offer to quit the dev server of the tab you are sitting
217
+ # in. An empty-but-successful answer is different: that is a desk with no
218
+ # windows left, which is exactly when orphans matter most.
219
+ if not terminal.terminal_status()["ok"]:
220
+ return _finish(projects)
221
+
222
+ # An orphan is a process nothing has a claim on any more. What can have
223
+ # one, and why each is a different fact with a different lifetime, is
224
+ # written down in model.Hold. Built here, most general first, so the most
225
+ # specific claim on a pid is the one that survives.
226
+ #
227
+ # A running application was free once: Terminal.app is a .app bundle, so
228
+ # the application rule covered the terminal too. That made the protection
229
+ # an accident of this driver rather than a property of the tool, and a
230
+ # driver whose emulator is a plain binary would have had none of it.
231
+ # terminal.owner_pids() states it instead.
232
+ #
233
+ # A multiplexer is not the driver's to answer, which is why it is here
234
+ # rather than behind that seam: a multiplexer is not the terminal you are
235
+ # in, it is a second one running inside it, and every driver inherits the
236
+ # same blind spot. What changed is which part of it holds anything. The
237
+ # server used to, which is far too much: a server outlives its panes, so
238
+ # everything it ever started stayed claimed forever. Measured: a
239
+ # `tmux run-shell -b` job hangs directly off the server with no pane at
240
+ # all, and was invisible to the sweep for as long as the server ran.
241
+ # A live pane holds its own subtree, and the server holds nothing unless
242
+ # it could not be asked.
243
+ holds: dict[int, Hold] = {}
244
+
245
+ def claim(hold: Hold) -> None:
246
+ """Record a claim. Which one survives is model.strongest()'s to say,
247
+ not the order these lines happen to be written in."""
248
+ holds[hold.pid] = strongest(holds.get(hold.pid), hold)
249
+
250
+ for r in rows:
251
+ if is_gui_app(r["cmd"]):
252
+ claim(Hold(r["pid"], "app", "part of a running app"))
253
+ for pid in terminal.owner_pids(rows):
254
+ claim(Hold(pid, "window", "open in a Terminal window"))
255
+ found = multiplexer.live_panes(rows)
256
+ for pid in found.unasked:
257
+ name = next((multiplexer_name(r["cmd"]) for r in rows
258
+ if r["pid"] == pid), "") or "a terminal multiplexer"
259
+ claim(Hold(pid, "multiplexer",
260
+ f"inside {name}, which could not say what it holds"))
261
+ for pid in found.asked:
262
+ # A server that answered keeps only what it is showing. Everything in
263
+ # a pane is claimed by that pane below; this covers what tmux draws
264
+ # without giving it a pane of its own, which is a popup, and it is
265
+ # told apart from a job the server merely runs by whether the server
266
+ # gave it a terminal. See model.Hold.
267
+ claim(Hold(pid, "shown", "shown by tmux", needs_tty=True))
268
+ for pane in found.live:
269
+ claim(Hold(pane.pid, "pane", f"open in {pane.where}"))
270
+ parent_of = {r["pid"]: r["ppid"] for r in rows}
271
+
272
+ # Only tree *roots* are orphans. Walking rows in ps order and claiming
273
+ # greedily made this depend on pid ordering: a child whose pid is lower
274
+ # than its parent's (they wrap around) got claimed first, and the parent
275
+ # was then skipped for overlapping. It reported a subtree and left the
276
+ # process that respawns it alive.
277
+ candidates = {}
278
+ for r in rows:
279
+ # An early out, not a guard. Everything a window accounted for is in
280
+ # claimed_pids along with its descendants, and the overlap check below
281
+ # refuses the same set again, so removing this line changes no answer.
282
+ # It is kept because skipping the work is cheaper than undoing it, and
283
+ # said out loud because a test cannot tell the two apart: this is one
284
+ # of the two lines here that no test can be written against, and
285
+ # pretending otherwise is worse than saying so.
286
+ if r["pid"] not in claimed_pids:
287
+ kind = classify_cmd(r["cmd"])
288
+ if not kind:
289
+ continue
290
+ # Asked once, read twice. Deciding "not an orphan" and reporting
291
+ # why are the same fact, and computing them separately is how
292
+ # under_a_root() and project_from_path() drifted apart.
293
+ hold = holder(r["pid"], holds, parent_of, on_a_tty=bool(r["tty"]))
294
+ if hold is None:
295
+ # Nothing was found on the way up, which is only an answer if
296
+ # the way up was there to walk. A chain that ran out of table
297
+ # or turned back on itself has not said "nobody has this", it
298
+ # has failed to look, and the difference matters because the
299
+ # word for a process nobody has is a target. Same rule as the
300
+ # multiplexer nobody could ask: not knowing protects.
301
+ if not traceable(r["pid"], parent_of):
302
+ continue
303
+ candidates[r["pid"]] = kind
304
+ elif held is not None and hold.kind in REPORTABLE_HOLDS:
305
+ held.append(hold)
306
+
307
+ def has_candidate_ancestor(pid: int) -> bool:
308
+ seen: set[int] = set()
309
+ pid = parent_of.get(pid, 0)
310
+ while pid and pid > 1 and pid not in seen:
311
+ if pid in candidates:
312
+ return True
313
+ seen.add(pid)
314
+ pid = parent_of.get(pid, 0)
315
+ return False
316
+
317
+ by_pid = {r["pid"]: r for r in rows}
318
+ # One lsof for every candidate, the way the tab loop above already does
319
+ # it. Asking per candidate is a subprocess each, with a 6s timeout each,
320
+ # on the hot path of every keypress that refreshes.
321
+ roots_of_candidates = [pid for pid in candidates
322
+ if not has_candidate_ancestor(pid)]
323
+ cwd_by_candidate = cwd_for_pids(roots_of_candidates)
324
+
325
+ for pid in roots_of_candidates:
326
+ kind = candidates[pid]
327
+ cmd = by_pid[pid]["cmd"]
328
+ # Where it is running first, what its arguments say second. The other
329
+ # order let any flag value that happened to contain a configured root
330
+ # (--config, --watch, a cache directory, a log path) decide which
331
+ # project a kill hits, and promote a process that its real working
332
+ # directory would have placed outside every root. Reading a project
333
+ # out of an argument is the mistake this codebase keeps finding; it
334
+ # stays only as the fallback for a process whose cwd cannot be read.
335
+ #
336
+ # An orphan belongs to a project, and no project means it is not
337
+ # offered. That used to be two questions with two different answers,
338
+ # "is it under a root" and "which project is it in", and a path could
339
+ # pass the first and be given a project by the second that does not
340
+ # exist. One lookup now, and None is a real answer.
341
+ project = index.of_path(cwd_by_candidate.get(pid)) or index.in_cmd(cmd)
342
+ if project is None:
343
+ continue
344
+ name, proj_path = project.name, project.path
345
+ pids = descendant_pids(pid, kids)
346
+ if pids & claimed_pids:
347
+ continue
348
+ mb = tree_mb(pids, rows, mem_sample)
349
+ # The floor is calibrated against phys_footprint, which is what `top`
350
+ # gives. Without that sample tree_mb falls back to `ps` RSS, and the
351
+ # floor sat *above* most real dev servers when it was measured against
352
+ # RSS: eleven orphaned astro servers holding 1.6 GB reported 1-2 MB
353
+ # each. So when the sample did not arrive the floor is not applied at
354
+ # all, because filtering on a number it does not describe is worse
355
+ # than not filtering. See config.ORPHAN_FLOOR_MB.
356
+ if mem_sample and mb < ORPHAN_FLOOR_MB:
357
+ continue
358
+ claimed_pids |= pids
359
+ proj = ensure(name, proj_path)
360
+ proj.sessions.append(Session(
361
+ kind="bg",
362
+ label=kind,
363
+ mb=mb,
364
+ pids=sorted(pids),
365
+ cmds={p: by_pid[p]["cmd"] for p in sorted(pids) if p in by_pid},
366
+ starts={p: by_pid[p]["started"] for p in sorted(pids)
367
+ if p in by_pid},
368
+ ))
369
+
370
+ return _finish(projects)
371
+
372
+
373
+ def _finish(projects: dict[str, Project]) -> list[Project]:
374
+ items = list(projects.values())
375
+ items.sort(key=lambda p: (-bool(p.window_ids), -p.mb, p.name))
376
+ for p in items:
377
+ p.sessions.sort(key=lambda s: (0 if s.kind == "window" else 1, -s.mb, s.label))
378
+ return items
379
+
380
+
381
+ def background_sessions(
382
+ project_name: str | None = None, *, projects: list[Project] | None = None
383
+ ) -> list[tuple[str, Session]]:
384
+ out: list[tuple[str, Session]] = []
385
+ for p in projects if projects is not None else build_projects():
386
+ # same_name, not ==: a project read off the disk and the same name
387
+ # typed at a prompt can differ by unicode normalisation alone.
388
+ if project_name is not None and not same_name(p.name, project_name):
389
+ continue
390
+ for s in p.sessions:
391
+ if s.kind == "bg":
392
+ out.append((p.name, s))
393
+ return out
394
+
395
+
396
+ def snapshot() -> dict:
397
+ held: list[Hold] = []
398
+ projects = build_projects(held=held)
399
+ # Read before list_profiles(), which is cheaper than the desk query and
400
+ # writes the same status. Asking afterwards reported the profile query's
401
+ # success in place of the desk query's failure.
402
+ status = terminal.terminal_status()
403
+ mem = procs.mem_summary()
404
+ tracked = sum(p.mb for p in projects)
405
+ return {
406
+ "mem": mem,
407
+ "terminal": status,
408
+ "tracked_mb": tracked,
409
+ "profiles": terminal.list_profiles(),
410
+ "projects": [p.to_dict() for p in projects if p.sessions],
411
+ # Why the list above is shorter than the machine, if it is.
412
+ "held": held_note(held),
413
+ }