focus-now 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,17 @@
1
+ Metadata-Version: 2.3
2
+ Name: focus-now
3
+ Version: 0.1.0
4
+ Summary: Add your description here
5
+ Author: Ian Atroshchenko
6
+ Author-email: Ian Atroshchenko <ianatroshchenko@gmail.com>
7
+ Requires-Dist: anthropic>=1.11.0
8
+ Requires-Dist: mss>=10.2.0
9
+ Requires-Dist: openai>=3.23.0
10
+ Requires-Dist: pillow>=12.3.0
11
+ Requires-Dist: psutil>=7.2.2
12
+ Requires-Dist: pyfiglet>=1.0.4
13
+ Requires-Dist: python-xlib>=0.33 ; sys_platform == 'linux'
14
+ Requires-Dist: typer>=0.27.2
15
+ Requires-Python: >=3.11
16
+ Description-Content-Type: text/markdown
17
+
File without changes
@@ -0,0 +1,30 @@
1
+ [project]
2
+ name = "focus-now"
3
+ version = "0.1.0"
4
+ description = "Add your description here"
5
+ readme = "README.md"
6
+ requires-python = ">=3.11"
7
+ dependencies = [
8
+ "anthropic>=1.11.0",
9
+ "mss>=10.2.0",
10
+ "openai>=3.23.0",
11
+ "pillow>=12.3.0",
12
+ "psutil>=7.2.2",
13
+ "pyfiglet>=1.0.4",
14
+ "python-xlib>=0.33 ; sys_platform == 'linux'",
15
+ "typer>=0.27.2",
16
+ ]
17
+
18
+ [[project.authors]]
19
+ name = "Ian Atroshchenko"
20
+ email = "ianatroshchenko@gmail.com"
21
+
22
+ [project.scripts]
23
+ focus = "focus.main:app"
24
+
25
+ [build-system]
26
+ requires = ["uv_build>=0.12.22,<0.13.0"]
27
+ build-backend = "uv_build"
28
+
29
+ [tool.uv.build-backend]
30
+ module-name = "focus"
@@ -0,0 +1,30 @@
1
+ [project]
2
+ name = "focus-now"
3
+ version = "0.1.0"
4
+ description = "Add your description here"
5
+ readme = "README.md"
6
+ authors = [
7
+ { name = "Ian Atroshchenko", email = "ianatroshchenko@gmail.com" }
8
+ ]
9
+ requires-python = ">=3.11"
10
+ dependencies = [
11
+ "anthropic>=1.11.0",
12
+ "mss>=10.2.0",
13
+ "openai>=3.23.0",
14
+ "pillow>=12.3.0",
15
+ "psutil>=7.2.2",
16
+ "pyfiglet>=1.0.4",
17
+ "python-xlib>=0.33 ; sys_platform == 'linux'",
18
+ "typer>=0.27.2",
19
+ ]
20
+
21
+ [project.scripts]
22
+ focus = "focus.main:app"
23
+
24
+ [build-system]
25
+ requires = ["uv_build>=0.12.22,<0.13.0"]
26
+ build-backend = "uv_build"
27
+
28
+ [tool.uv.build-backend]
29
+ module-name = "focus"
30
+
File without changes
File without changes
@@ -0,0 +1,331 @@
1
+ import json
2
+ import time
3
+ from abc import ABC, abstractmethod
4
+ from collections import deque
5
+ from dataclasses import dataclass, field
6
+ from typing import Any
7
+
8
+ import anthropic
9
+ import openai
10
+
11
+ from focus.tools import TOOLS, Screenshot, ToolResult, run_tool
12
+ from focus.utils.log import err, log, warn
13
+ from focus.utils.prompts import build_system_prompt, check_in
14
+
15
+
16
+ @dataclass
17
+ class ToolCall:
18
+ id: str
19
+ name: str
20
+ args: dict[str, Any]
21
+ # set when the model's arguments couldn't be parsed; the call is reported back, not run
22
+ error: str | None = None
23
+
24
+
25
+ @dataclass
26
+ class Turn:
27
+ """one model response, normalized across providers"""
28
+
29
+ text: str
30
+ tool_calls: list[ToolCall] = field(default_factory=list)
31
+ stopped_early: str | None = None
32
+
33
+
34
+ class BaseAgent(ABC):
35
+ DEFAULT_MODEL: str
36
+ # API errors that retrying won't fix: bad key, unknown model, malformed request
37
+ FATAL_ERRORS: tuple[type[Exception], ...] = ()
38
+
39
+ def __init__(
40
+ self,
41
+ model: str,
42
+ api_key: str,
43
+ task: str | None = None,
44
+ interval: float = 20.0,
45
+ effort: str | None = "low",
46
+ max_turns: int = 10,
47
+ max_notes: int = 12,
48
+ ):
49
+ self.model = model
50
+ self.api_key = api_key
51
+ self.system_prompt = build_system_prompt(task)
52
+ self.interval = interval
53
+ self.effort = effort
54
+ self.max_turns = max_turns
55
+ self.notes: deque[str] = deque(maxlen=max_notes)
56
+ self.started_at = time.time()
57
+
58
+ @abstractmethod
59
+ def _user_message(self, text: str, images: list[Screenshot]) -> list[Any]:
60
+ """conversation items for a user message"""
61
+
62
+ @abstractmethod
63
+ def _complete(self, conversation: list[Any]) -> Turn:
64
+ """call the model, append its output to the conversation, and parse the turn"""
65
+
66
+ @abstractmethod
67
+ def _tool_results(self, results: list[tuple[ToolCall, ToolResult]]) -> list[Any]:
68
+ """conversation items reporting tool results back to the model"""
69
+
70
+ def capture_screen(self) -> ToolResult:
71
+ return run_tool("capture_screen", {})
72
+
73
+ def use_tool(self, call: ToolCall) -> ToolResult:
74
+ if call.error:
75
+ result = ToolResult(call.error, is_error=True)
76
+ else:
77
+ result = run_tool(call.name, call.args)
78
+
79
+ args = ", ".join(f"{k}={v!r}" for k, v in call.args.items())
80
+ line = f"[cyan]{call.name}[/cyan]({args})"
81
+ if result.is_error:
82
+ warn(f"{line} -> {result.text}")
83
+ else:
84
+ log(line)
85
+ return result
86
+
87
+ def prompt(self, text: str, images: list[Screenshot] | None = None) -> str:
88
+ """send the model a prompt and let it use tools until it's done. returns its final text"""
89
+ conversation = self._user_message(text, images or [])
90
+
91
+ for _ in range(self.max_turns):
92
+ turn = self._complete(conversation)
93
+ if turn.stopped_early:
94
+ warn(turn.stopped_early)
95
+ if turn.stopped_early or not turn.tool_calls:
96
+ return turn.text
97
+
98
+ results = [(call, self.use_tool(call)) for call in turn.tool_calls]
99
+ conversation.extend(self._tool_results(results))
100
+
101
+ warn(f"check-in hit the {self.max_turns} turn limit")
102
+ return turn.text
103
+
104
+ def check_in(self) -> str:
105
+ notes = "\n".join(f"- {note}" for note in self.notes)
106
+ text = check_in.format(
107
+ time=time.strftime("%H:%M:%S"),
108
+ elapsed=_format_duration(time.time() - self.started_at),
109
+ notes=notes or "(none yet, this is the first check-in)",
110
+ )
111
+ screen = self.capture_screen()
112
+ if screen.is_error:
113
+ text += f"\n\n(automatic screenshot failed: {screen.text})"
114
+
115
+ summary = self.prompt(text, screen.images).strip()
116
+ if summary:
117
+ # the model's closing line is its memory for the next check-in
118
+ self.notes.append(f"{time.strftime('%H:%M')} {summary.splitlines()[-1]}")
119
+ log(summary)
120
+ return summary
121
+
122
+ def run(self) -> None:
123
+ log(f"monitoring every {self.interval:g}s with {self.model}. ctrl+c to stop")
124
+ try:
125
+ while True:
126
+ try:
127
+ self.check_in()
128
+ except self.FATAL_ERRORS as e:
129
+ err(f"{type(e).__name__}: {e}")
130
+ except Exception as e:
131
+ warn(f"check-in failed, retrying next interval: {type(e).__name__}: {e}")
132
+ time.sleep(self.interval)
133
+ except KeyboardInterrupt:
134
+ log("focus session ended. nice work :)")
135
+
136
+
137
+ class AnthropicAgent(BaseAgent):
138
+ DEFAULT_MODEL = "claude-sonnet-5-5"
139
+ FATAL_ERRORS = (
140
+ anthropic.AuthenticationError,
141
+ anthropic.PermissionDeniedError,
142
+ anthropic.NotFoundError,
143
+ anthropic.BadRequestError,
144
+ )
145
+ # models that accept the server-side refusal fallback. screen contents are
146
+ # arbitrary, so a safety decline on one frame shouldn't end the check-in
147
+ FALLBACK_MODELS = {"claude-fable-5-1", "claude-opus-5-5", "claude-opus-5", "claude-sonnet-5-5"}
148
+ MAX_TOKENS = 16000
149
+
150
+ def __init__(self, model: str = DEFAULT_MODEL, **kwargs):
151
+ super().__init__(model=model, **kwargs)
152
+ self.client = anthropic.Anthropic(api_key=self.api_key)
153
+ self.tools = [
154
+ {"name": t.name, "description": t.description, "input_schema": t.parameters}
155
+ for t in TOOLS
156
+ ]
157
+
158
+ self.request_options: dict[str, Any] = {}
159
+ if self.effort and not model.startswith("claude-haiku"):
160
+ self.request_options["output_config"] = {"effort": self.effort}
161
+
162
+ self.create_message = self.client.messages.create
163
+ if model in self.FALLBACK_MODELS:
164
+ self.create_message = self.client.beta.messages.create
165
+ self.request_options |= {
166
+ "betas": ["server-side-fallback-2026-07-01"],
167
+ "fallbacks": "default",
168
+ }
169
+
170
+ def _user_message(self, text, images):
171
+ content = [self._image(shot) for shot in images]
172
+ content.append({"type": "text", "text": text})
173
+ return [{"role": "user", "content": content}]
174
+
175
+ def _complete(self, conversation):
176
+ response = self.create_message(
177
+ model=self.model,
178
+ max_tokens=self.MAX_TOKENS,
179
+ # system and tools never change, so this prefix stays cached across check-ins;
180
+ # the top-level cache_control caches the growing conversation within one
181
+ system=[
182
+ {"type": "text", "text": self.system_prompt, "cache_control": {"type": "ephemeral"}}
183
+ ],
184
+ cache_control={"type": "ephemeral"},
185
+ tools=self.tools,
186
+ messages=conversation,
187
+ **self.request_options,
188
+ )
189
+ # append the content unchanged; thinking and fallback blocks must be sent back as-is
190
+ conversation.append({"role": "assistant", "content": response.content})
191
+
192
+ return Turn(
193
+ text="\n".join(b.text for b in response.content if b.type == "text"),
194
+ tool_calls=[
195
+ ToolCall(b.id, b.name, dict(b.input))
196
+ for b in response.content
197
+ if b.type == "tool_use"
198
+ ],
199
+ stopped_early={
200
+ "refusal": "the model declined to review this screen",
201
+ "max_tokens": "the model ran out of output tokens mid-check-in",
202
+ }.get(response.stop_reason),
203
+ )
204
+
205
+ def _tool_results(self, results):
206
+ blocks = [
207
+ {
208
+ "type": "tool_result",
209
+ "tool_use_id": call.id,
210
+ "content": [
211
+ {"type": "text", "text": result.text},
212
+ *(self._image(shot) for shot in result.images),
213
+ ],
214
+ "is_error": result.is_error,
215
+ }
216
+ for call, result in results
217
+ ]
218
+ # every result for a turn goes back in one user message
219
+ return [{"role": "user", "content": blocks}]
220
+
221
+ @staticmethod
222
+ def _image(shot: Screenshot) -> dict[str, Any]:
223
+ return {
224
+ "type": "image",
225
+ "source": {"type": "base64", "media_type": shot.media_type, "data": shot.b64},
226
+ }
227
+
228
+
229
+ class OpenAIAgent(BaseAgent):
230
+ DEFAULT_MODEL = "gpt-5"
231
+ FATAL_ERRORS = (
232
+ openai.AuthenticationError,
233
+ openai.PermissionDeniedError,
234
+ openai.NotFoundError,
235
+ openai.BadRequestError,
236
+ )
237
+
238
+ def __init__(self, model: str = DEFAULT_MODEL, **kwargs):
239
+ super().__init__(model=model, **kwargs)
240
+ self.client = openai.OpenAI(api_key=self.api_key)
241
+ self.tools = [
242
+ {
243
+ "type": "function",
244
+ "name": t.name,
245
+ "description": t.description,
246
+ "parameters": t.parameters,
247
+ "strict": False,
248
+ }
249
+ for t in TOOLS
250
+ ]
251
+
252
+ self.request_options: dict[str, Any] = {}
253
+ if not model.startswith(("gpt-4", "gpt-3")):
254
+ # with store=False, reasoning carries over between turns as encrypted items
255
+ self.request_options["include"] = ["reasoning.encrypted_content"]
256
+ if self.effort:
257
+ self.request_options["reasoning"] = {"effort": self.effort}
258
+
259
+ def _user_message(self, text, images):
260
+ content = [self._image(shot) for shot in images]
261
+ content.append({"type": "input_text", "text": text})
262
+ return [{"role": "user", "content": content}]
263
+
264
+ def _complete(self, conversation):
265
+ response = self.client.responses.create(
266
+ model=self.model,
267
+ instructions=self.system_prompt,
268
+ input=conversation,
269
+ tools=self.tools,
270
+ # screenshots are private, so nothing is kept server side
271
+ store=False,
272
+ **self.request_options,
273
+ )
274
+ conversation.extend(response.output)
275
+
276
+ stopped_early = None
277
+ if response.status == "incomplete":
278
+ reason = getattr(response.incomplete_details, "reason", "unknown")
279
+ stopped_early = f"the model stopped early ({reason})"
280
+
281
+ return Turn(
282
+ text=response.output_text,
283
+ tool_calls=[
284
+ self._parse_call(item) for item in response.output if item.type == "function_call"
285
+ ],
286
+ stopped_early=stopped_early,
287
+ )
288
+
289
+ def _tool_results(self, results):
290
+ return [
291
+ {"type": "function_call_output", "call_id": call.id, "output": self._output(result)}
292
+ for call, result in results
293
+ ]
294
+
295
+ @staticmethod
296
+ def _parse_call(item) -> ToolCall:
297
+ try:
298
+ return ToolCall(item.call_id, item.name, json.loads(item.arguments or "{}"))
299
+ except json.JSONDecodeError:
300
+ return ToolCall(item.call_id, item.name, {}, error="tool arguments were not valid JSON")
301
+
302
+ @classmethod
303
+ def _output(cls, result: ToolResult) -> str | list[dict[str, Any]]:
304
+ text = f"error: {result.text}" if result.is_error else result.text
305
+ if not result.images:
306
+ return text
307
+ return [{"type": "input_text", "text": text}, *map(cls._image, result.images)]
308
+
309
+ @staticmethod
310
+ def _image(shot: Screenshot) -> dict[str, Any]:
311
+ return {"type": "input_image", "image_url": shot.data_url, "detail": "high"}
312
+
313
+
314
+ AGENTS: dict[str, type[BaseAgent]] = {
315
+ "anthropic": AnthropicAgent,
316
+ "openai": OpenAIAgent,
317
+ }
318
+ DEFAULT_MODELS = {provider: agent.DEFAULT_MODEL for provider, agent in AGENTS.items()}
319
+
320
+
321
+ def create_agent(provider: str, model: str | None = None, **kwargs) -> BaseAgent:
322
+ if provider not in AGENTS:
323
+ err(f"unknown provider '{provider}'. choose one of: {', '.join(AGENTS)}")
324
+ agent = AGENTS[provider]
325
+ return agent(model=model or agent.DEFAULT_MODEL, **kwargs)
326
+
327
+
328
+ def _format_duration(seconds: float) -> str:
329
+ minutes, seconds = divmod(int(seconds), 60)
330
+ hours, minutes = divmod(minutes, 60)
331
+ return f"{hours}h {minutes}m" if hours else f"{minutes}m {seconds}s"
@@ -0,0 +1,62 @@
1
+ import typer
2
+ from focus.agent.agent import DEFAULT_MODELS, create_agent
3
+ from focus.utils.log import warn, err, log
4
+
5
+ import os
6
+ from pyfiglet import figlet_format
7
+
8
+ app = typer.Typer()
9
+
10
+
11
+ @app.callback(invoke_without_command=True)
12
+ def main(
13
+ ctx: typer.Context,
14
+ key: str | None = typer.Option(None, "--key", help="api key to use for Focus"),
15
+ provider: str = typer.Option(
16
+ "anthropic", "--provider", help="model provider to use for Focus (anthropic, openai)"
17
+ ),
18
+ model: str | None = typer.Option(
19
+ None,
20
+ "--model",
21
+ help=f"model type to use for Focus (defaults: {', '.join(f'{p}={m}' for p, m in DEFAULT_MODELS.items())})",
22
+ ),
23
+ task: str | None = typer.Option(
24
+ None,
25
+ "--task",
26
+ help="task you wish to complete in this Focus session. without one, Focus keeps you on generally productive work",
27
+ ),
28
+ interval: float = typer.Option(
29
+ 20.0, "--interval", help="seconds to wait between screen check-ins"
30
+ ),
31
+ effort: str = typer.Option(
32
+ "low", "--effort", help="how hard the model thinks per check-in (low, medium, high)"
33
+ ),
34
+ ):
35
+ if ctx.invoked_subcommand is None:
36
+ start_focus(provider, model, key, task, interval, effort)
37
+
38
+
39
+ def start_focus(
40
+ provider: str,
41
+ model: str | None,
42
+ key: str | None = None,
43
+ task: str | None = None,
44
+ interval: float = 20.0,
45
+ effort: str = "low",
46
+ ):
47
+ if key == None:
48
+ warn("provider api key not given. searching...")
49
+ key = os.getenv(provider.upper() + "_API_KEY")
50
+
51
+ if key == None:
52
+ err(
53
+ "provider api key not found. please set $OPENAI_API_KEY, $ANTHROPIC_API_KEY, or pass in api_key with the [green]--key[/green] argument"
54
+ )
55
+
56
+ f = figlet_format("FOCUS!", font="slant")
57
+ print(f)
58
+ agent = create_agent(
59
+ provider, model, api_key=key, task=task, interval=interval, effort=effort
60
+ )
61
+ log("focus is now running :)")
62
+ agent.run()
@@ -0,0 +1,4 @@
1
+ from focus.tools.base import Screenshot, Tool, ToolResult
2
+ from focus.tools.definitions import TOOLS, run_tool
3
+
4
+ __all__ = ["TOOLS", "Screenshot", "Tool", "ToolResult", "run_tool"]
@@ -0,0 +1,70 @@
1
+ import base64
2
+ import io
3
+ from dataclasses import dataclass, field
4
+ from typing import Any, Callable
5
+
6
+ from PIL import Image
7
+
8
+ # providers downscale anything larger, so bigger images only cost bandwidth and tokens
9
+ MAX_IMAGE_EDGE = 1568
10
+ JPEG_QUALITY = 75
11
+
12
+
13
+ @dataclass
14
+ class Screenshot:
15
+ data: bytes
16
+ media_type: str
17
+ width: int
18
+ height: int
19
+ label: str = ""
20
+
21
+ @classmethod
22
+ def from_image(cls, image: Image.Image, label: str = "") -> "Screenshot":
23
+ image.thumbnail((MAX_IMAGE_EDGE, MAX_IMAGE_EDGE))
24
+ if image.mode != "RGB":
25
+ image = image.convert("RGB")
26
+ buffer = io.BytesIO()
27
+ image.save(buffer, format="JPEG", quality=JPEG_QUALITY)
28
+ return cls(buffer.getvalue(), "image/jpeg", image.width, image.height, label)
29
+
30
+ @property
31
+ def b64(self) -> str:
32
+ return base64.standard_b64encode(self.data).decode("ascii")
33
+
34
+ @property
35
+ def data_url(self) -> str:
36
+ return f"data:{self.media_type};base64,{self.b64}"
37
+
38
+
39
+ @dataclass
40
+ class ToolResult:
41
+ text: str
42
+ images: list[Screenshot] = field(default_factory=list)
43
+ is_error: bool = False
44
+
45
+
46
+ @dataclass
47
+ class Tool:
48
+ name: str
49
+ description: str
50
+ parameters: dict[str, Any]
51
+ handler: Callable[..., ToolResult]
52
+
53
+
54
+ def tool(description: str, parameters: dict[str, Any] | None = None):
55
+ """turn a handler function into a model-facing Tool named after the function"""
56
+
57
+ def wrap(handler: Callable[..., ToolResult]) -> Tool:
58
+ return Tool(handler.__name__, description, parameters or schema(), handler)
59
+
60
+ return wrap
61
+
62
+
63
+ def schema(required: tuple[str, ...] = (), **properties: dict[str, Any]) -> dict[str, Any]:
64
+ """JSON schema for a tool's arguments"""
65
+ return {
66
+ "type": "object",
67
+ "properties": properties,
68
+ "required": list(required),
69
+ "additionalProperties": False,
70
+ }
@@ -0,0 +1,149 @@
1
+ import time
2
+ from typing import Any
3
+
4
+ from focus.tools.base import Screenshot, Tool, ToolResult, schema, tool
5
+ from focus.tools.platform import PlatformError, get_platform
6
+
7
+ MAX_WAIT_SECONDS = 60
8
+
9
+
10
+ @tool(
11
+ "Take a screenshot of every monitor and return the images. This is your primary way "
12
+ "of seeing what the user is doing. Call it again after any action (closing a tab, "
13
+ "killing an app) to confirm the result."
14
+ )
15
+ def capture_screen() -> ToolResult:
16
+ shots = [
17
+ Screenshot.from_image(image, label=f"monitor {i}")
18
+ for i, image in enumerate(get_platform().grab_screens(), start=1)
19
+ ]
20
+ sizes = ", ".join(f"{s.label}: {s.width}x{s.height}" for s in shots)
21
+ return ToolResult(
22
+ f"captured {len(shots)} screen(s) at {time.strftime('%H:%M:%S')} ({sizes})",
23
+ images=shots,
24
+ )
25
+
26
+
27
+ @tool(
28
+ "List the user's open windows with their window id, app name, process id, and title. "
29
+ "The focused window is marked with '*'. Browser window titles usually contain the "
30
+ "active tab's page title, which helps tell productive browsing from distraction. Use "
31
+ "the ids/pids/app names here for close_tab and kill_app."
32
+ )
33
+ def list_windows() -> ToolResult:
34
+ windows = get_platform().list_windows()
35
+ if not windows:
36
+ return ToolResult("no open windows found")
37
+
38
+ lines = [
39
+ f"{'*' if w.focused else ' '} id={w.id} app={w.app} pid={w.pid} title={w.title!r}"
40
+ for w in windows
41
+ ]
42
+ return ToolResult("open windows (* = focused):\n" + "\n".join(lines))
43
+
44
+
45
+ @tool(
46
+ "Show the user a desktop notification. This is the gentlest intervention: use it "
47
+ "first to nudge the user back on task before taking stronger action. Keep messages "
48
+ "short, specific, and friendly.",
49
+ schema(
50
+ required=("title", "message"),
51
+ title={"type": "string", "description": "short notification title"},
52
+ message={"type": "string", "description": "notification body, one or two sentences"},
53
+ urgency={
54
+ "type": "string",
55
+ "enum": ["low", "normal", "critical"],
56
+ "description": "critical notifications stay on screen until dismissed",
57
+ },
58
+ ),
59
+ )
60
+ def notify_user(title: str, message: str, urgency: str = "normal") -> ToolResult:
61
+ get_platform().notify(title, message, urgency)
62
+ return ToolResult(f"notification shown: {title!r}")
63
+
64
+
65
+ @tool(
66
+ "Close the active tab of a browser (or other tabbed app) by focusing the window and "
67
+ "pressing the close-tab keyboard shortcut. Only use this when the distracting content "
68
+ "is the ACTIVE tab of that window, as seen in a screenshot or the window title. In "
69
+ "non-tabbed apps the same shortcut may close the whole window or document, so don't "
70
+ "use it there.",
71
+ schema(
72
+ window_id={
73
+ "type": "string",
74
+ "description": "window id from list_windows. omit to target the focused window",
75
+ },
76
+ ),
77
+ )
78
+ def close_tab(window_id: str | None = None) -> ToolResult:
79
+ get_platform().close_tab(window_id)
80
+ target = f"window {window_id}" if window_id else "the focused window"
81
+ return ToolResult(
82
+ f"sent the close-tab shortcut to {target}. capture the screen to confirm it worked."
83
+ )
84
+
85
+
86
+ @tool(
87
+ "Terminate an application by process name (kills every process with that exact name, "
88
+ "e.g. 'Discord') or by pid. This is the strongest intervention; reserve it for clear, "
89
+ "repeated distraction after the user has already been warned. System processes, Focus "
90
+ "itself, and the terminal it runs in are protected and will be refused.",
91
+ schema(
92
+ name={"type": "string", "description": "exact process name, case-insensitive"},
93
+ pid={"type": "integer", "description": "a specific process id"},
94
+ ),
95
+ )
96
+ def kill_app(name: str | None = None, pid: int | None = None) -> ToolResult:
97
+ report = get_platform().kill_app(name=name, pid=pid)
98
+ if not report.killed and not report.refused:
99
+ return ToolResult(
100
+ f"no running process matched {name or pid}. "
101
+ "use list_windows to find the right app name or pid.",
102
+ is_error=True,
103
+ )
104
+
105
+ lines = []
106
+ if report.killed:
107
+ lines.append("killed: " + ", ".join(report.killed))
108
+ if report.refused:
109
+ lines.append(
110
+ "refused (protected system process, Focus itself, or another user's): "
111
+ + ", ".join(report.refused)
112
+ )
113
+ return ToolResult("\n".join(lines), is_error=not report.killed)
114
+
115
+
116
+ @tool(
117
+ f"Pause for a few seconds (1-{MAX_WAIT_SECONDS}) before looking again. Use this when "
118
+ "you suspect distraction but aren't sure yet, e.g. the user might be briefly checking "
119
+ "a message or looking something up, then capture_screen again to see whether it "
120
+ "continued.",
121
+ schema(
122
+ required=("seconds",),
123
+ seconds={"type": "number", "minimum": 1, "maximum": MAX_WAIT_SECONDS},
124
+ ),
125
+ )
126
+ def wait(seconds: float) -> ToolResult:
127
+ seconds = max(1.0, min(float(seconds), MAX_WAIT_SECONDS))
128
+ time.sleep(seconds)
129
+ return ToolResult(f"waited {seconds:g} seconds")
130
+
131
+
132
+ TOOLS: list[Tool] = [capture_screen, list_windows, notify_user, close_tab, kill_app, wait]
133
+ TOOLS_BY_NAME: dict[str, Tool] = {t.name: t for t in TOOLS}
134
+
135
+
136
+ def run_tool(name: str, args: dict[str, Any]) -> ToolResult:
137
+ """run a tool call from any provider. failures become error results the model can
138
+ read and react to, rather than exceptions that end the check-in"""
139
+ if name not in TOOLS_BY_NAME:
140
+ return ToolResult(f"unknown tool: {name}", is_error=True)
141
+
142
+ try:
143
+ return TOOLS_BY_NAME[name].handler(**args)
144
+ except PlatformError as e:
145
+ return ToolResult(str(e), is_error=True)
146
+ except TypeError as e:
147
+ return ToolResult(f"bad arguments for {name}: {e}", is_error=True)
148
+ except Exception as e:
149
+ return ToolResult(f"{name} failed: {type(e).__name__}: {e}", is_error=True)
@@ -0,0 +1,40 @@
1
+ import os
2
+ import sys
3
+ from functools import cache
4
+
5
+ from focus.tools.platform.base import (
6
+ KillReport,
7
+ Platform,
8
+ PlatformError,
9
+ UnsupportedPlatform,
10
+ WindowInfo,
11
+ )
12
+
13
+
14
+ @cache
15
+ def get_platform() -> Platform:
16
+ """the Platform implementation for the OS and display server Focus is running on"""
17
+ if sys.platform.startswith("linux"):
18
+ if os.environ.get("WAYLAND_DISPLAY") or os.environ.get("XDG_SESSION_TYPE") == "wayland":
19
+ from focus.tools.platform.linux import WaylandPlatform
20
+
21
+ return WaylandPlatform()
22
+
23
+ from focus.tools.platform.linux import X11Platform
24
+
25
+ return X11Platform()
26
+
27
+ if sys.platform == "darwin":
28
+ from focus.tools.platform.macos import MacOSPlatform
29
+
30
+ return MacOSPlatform()
31
+
32
+ if sys.platform == "win32":
33
+ from focus.tools.platform.windows import WindowsPlatform
34
+
35
+ return WindowsPlatform()
36
+
37
+ return UnsupportedPlatform(sys.platform)
38
+
39
+
40
+ __all__ = ["get_platform", "KillReport", "Platform", "PlatformError", "WindowInfo"]
@@ -0,0 +1,118 @@
1
+ from abc import ABC, abstractmethod
2
+ from contextlib import suppress
3
+ from dataclasses import dataclass
4
+
5
+ import psutil
6
+ from PIL import Image
7
+
8
+
9
+ class PlatformError(Exception):
10
+ """raised when the OS can't do what was asked (missing utility, no display, ...)"""
11
+
12
+
13
+ @dataclass
14
+ class WindowInfo:
15
+ id: str
16
+ title: str
17
+ app: str
18
+ pid: int | None
19
+ focused: bool = False
20
+
21
+
22
+ @dataclass
23
+ class KillReport:
24
+ killed: list[str]
25
+ refused: list[str]
26
+
27
+
28
+ class Platform(ABC):
29
+ name: str = "unknown"
30
+
31
+ # process names Focus must never kill, whatever the model asks. each OS fills this in
32
+ protected_processes: frozenset[str] = frozenset()
33
+
34
+ @abstractmethod
35
+ def grab_screens(self) -> list[Image.Image]:
36
+ """one image per monitor"""
37
+
38
+ @abstractmethod
39
+ def list_windows(self) -> list[WindowInfo]:
40
+ """open top-level windows, with the focused one marked"""
41
+
42
+ @abstractmethod
43
+ def notify(self, title: str, message: str, urgency: str = "normal") -> None:
44
+ """show a system notification. urgency is one of low/normal/critical"""
45
+
46
+ @abstractmethod
47
+ def focus_window(self, window_id: str) -> None:
48
+ """raise and focus a window by the id returned from list_windows"""
49
+
50
+ @abstractmethod
51
+ def send_close_tab_shortcut(self) -> None:
52
+ """press the OS's close-tab shortcut (ctrl+w, cmd+w, ...) in the focused window"""
53
+
54
+ def close_tab(self, window_id: str | None = None) -> None:
55
+ if window_id is not None:
56
+ self.focus_window(window_id)
57
+ self.send_close_tab_shortcut()
58
+
59
+ def kill_app(self, name: str | None = None, pid: int | None = None) -> KillReport:
60
+ if name is None and pid is None:
61
+ raise PlatformError("give either an app name or a pid")
62
+
63
+ me = psutil.Process()
64
+ user = me.username()
65
+ protected_pids = {me.pid, *(parent.pid for parent in me.parents())}
66
+ targets, refused = [], []
67
+
68
+ for proc in psutil.process_iter(["pid", "name", "username"]):
69
+ proc_pid, proc_name = proc.info["pid"], (proc.info["name"] or "").lower()
70
+ matches = proc_pid == pid or (name is not None and proc_name == name.lower())
71
+ if not matches:
72
+ continue
73
+
74
+ is_protected = (
75
+ proc_pid in protected_pids
76
+ or proc_name in self.protected_processes
77
+ or proc.info["username"] != user
78
+ )
79
+ (refused if is_protected else targets).append(proc)
80
+
81
+ for proc in targets:
82
+ with suppress(psutil.NoSuchProcess):
83
+ proc.terminate()
84
+ _, still_alive = psutil.wait_procs(targets, timeout=3)
85
+ for proc in still_alive:
86
+ with suppress(psutil.NoSuchProcess):
87
+ proc.kill()
88
+
89
+ return KillReport(killed=_labels(targets), refused=_labels(refused))
90
+
91
+
92
+ class UnsupportedPlatform(Platform):
93
+ """placeholder for OSes that don't have an implementation yet"""
94
+
95
+ def __init__(self, os_name: str):
96
+ self.name = os_name
97
+
98
+ def _unsupported(self):
99
+ raise PlatformError(f"{self.name} is not supported by Focus yet")
100
+
101
+ def grab_screens(self):
102
+ self._unsupported()
103
+
104
+ def list_windows(self):
105
+ self._unsupported()
106
+
107
+ def notify(self, title, message, urgency="normal"):
108
+ self._unsupported()
109
+
110
+ def focus_window(self, window_id):
111
+ self._unsupported()
112
+
113
+ def send_close_tab_shortcut(self):
114
+ self._unsupported()
115
+
116
+
117
+ def _labels(procs: list[psutil.Process]) -> list[str]:
118
+ return [f"{p.info['name']} (pid {p.info['pid']})" for p in procs]
@@ -0,0 +1,157 @@
1
+ import io
2
+ import shutil
3
+ import subprocess
4
+ import time
5
+
6
+ import mss
7
+ from PIL import Image
8
+ from Xlib import XK, X, display
9
+ from Xlib.error import XError
10
+ from Xlib.ext import xtest
11
+ from Xlib.protocol import event
12
+
13
+ from focus.tools.platform.base import Platform, PlatformError, WindowInfo
14
+
15
+
16
+ class LinuxPlatform(Platform):
17
+ """behaviour shared by every linux display server"""
18
+
19
+ name = "linux"
20
+
21
+ protected_processes = frozenset(
22
+ # core system services
23
+ "systemd init xorg x xwayland dbus-daemon dbus-broker sshd login sddm gdm lightdm "
24
+ "pipewire pipewire-pulse wireplumber pulseaudio "
25
+ # window managers, compositors, desktop shells and bars
26
+ "i3 i3bar sway hyprland gnome-shell kwin_x11 kwin_wayland plasmashell xfwm4 "
27
+ "openbox bspwm awesome polybar waybar".split()
28
+ )
29
+
30
+ def notify(self, title: str, message: str, urgency: str = "normal") -> None:
31
+ self._require("notify-send", "notifications need `notify-send` (libnotify)")
32
+ self._run("notify-send", "-a", "Focus", "-u", urgency, title, message)
33
+
34
+ @staticmethod
35
+ def _require(command: str, reason: str) -> None:
36
+ if not shutil.which(command):
37
+ raise PlatformError(f"{reason} installed")
38
+
39
+ @staticmethod
40
+ def _run(*cmd: str) -> bytes:
41
+ try:
42
+ return subprocess.run(cmd, check=True, capture_output=True, timeout=10).stdout
43
+ except subprocess.CalledProcessError as e:
44
+ stderr = e.stderr.decode(errors="replace").strip()
45
+ raise PlatformError(f"`{cmd[0]}` failed: {stderr}") from e
46
+ except subprocess.TimeoutExpired as e:
47
+ raise PlatformError(f"`{cmd[0]}` timed out") from e
48
+
49
+
50
+ class X11Platform(LinuxPlatform):
51
+ """screenshots via mss, window management via EWMH, keystrokes via XTest"""
52
+
53
+ def __init__(self):
54
+ try:
55
+ self._display = display.Display()
56
+ self._screen_grabber = mss.MSS()
57
+ except Exception as e:
58
+ raise PlatformError(f"couldn't connect to the X server: {e}") from e
59
+ self._root = self._display.screen().root
60
+
61
+ def grab_screens(self) -> list[Image.Image]:
62
+ # monitors[0] is every monitor combined; the rest are the individual monitors
63
+ monitors = self._screen_grabber.monitors
64
+ return [
65
+ Image.frombytes("RGB", shot.size, shot.rgb)
66
+ for shot in map(self._screen_grabber.grab, monitors[1:] or monitors[:1])
67
+ ]
68
+
69
+ def list_windows(self) -> list[WindowInfo]:
70
+ client_ids = self._root_property("_NET_CLIENT_LIST")
71
+ active_ids = self._root_property("_NET_ACTIVE_WINDOW")
72
+ active_id = active_ids[0] if active_ids else None
73
+
74
+ windows = []
75
+ for window_id in client_ids:
76
+ try:
77
+ windows.append(self._window_info(window_id, focused=window_id == active_id))
78
+ except XError:
79
+ # the window closed between listing and inspecting it
80
+ continue
81
+ return windows
82
+
83
+ def focus_window(self, window_id: str) -> None:
84
+ window = self._display.create_resource_object("window", int(window_id, 16))
85
+ # source indication 2 ("pager") is honoured more readily by window managers
86
+ request = event.ClientMessage(
87
+ window=window,
88
+ client_type=self._atom("_NET_ACTIVE_WINDOW"),
89
+ data=(32, [2, X.CurrentTime, 0, 0, 0]),
90
+ )
91
+ self._root.send_event(
92
+ request, event_mask=X.SubstructureRedirectMask | X.SubstructureNotifyMask
93
+ )
94
+ self._display.sync()
95
+ # let the window manager finish switching before keys are sent to the window
96
+ time.sleep(0.25)
97
+
98
+ def send_close_tab_shortcut(self) -> None:
99
+ ctrl = self._display.keysym_to_keycode(XK.XK_Control_L)
100
+ w = self._display.keysym_to_keycode(XK.string_to_keysym("w"))
101
+ for event_type, key in [
102
+ (X.KeyPress, ctrl),
103
+ (X.KeyPress, w),
104
+ (X.KeyRelease, w),
105
+ (X.KeyRelease, ctrl),
106
+ ]:
107
+ xtest.fake_input(self._display, event_type, key)
108
+ self._display.sync()
109
+
110
+ def _window_info(self, window_id: int, focused: bool) -> WindowInfo:
111
+ window = self._display.create_resource_object("window", window_id)
112
+ title = window.get_full_text_property(
113
+ self._atom("_NET_WM_NAME"), self._atom("UTF8_STRING")
114
+ )
115
+ wm_class = window.get_wm_class()
116
+ pid = window.get_full_property(self._atom("_NET_WM_PID"), X.AnyPropertyType)
117
+ return WindowInfo(
118
+ id=hex(window_id),
119
+ title=title or window.get_wm_name() or "",
120
+ app=wm_class[1] if wm_class else "unknown",
121
+ pid=int(pid.value[0]) if pid else None,
122
+ focused=focused,
123
+ )
124
+
125
+ def _root_property(self, name: str) -> list[int]:
126
+ prop = self._root.get_full_property(self._atom(name), X.AnyPropertyType)
127
+ return list(prop.value) if prop else []
128
+
129
+ def _atom(self, name: str) -> int:
130
+ # get_atom caches, unlike intern_atom which asks the X server every time
131
+ return self._display.get_atom(name)
132
+
133
+
134
+ class WaylandPlatform(LinuxPlatform):
135
+ """
136
+ wayland doesn't let clients inspect or control each other's windows, so this
137
+ backend relies on compositor utilities and can't list or focus windows
138
+ """
139
+
140
+ def grab_screens(self) -> list[Image.Image]:
141
+ self._require("grim", "screen capture on wayland needs `grim` (wlroots compositors)")
142
+ return [Image.open(io.BytesIO(self._run("grim", "-t", "png", "-")))]
143
+
144
+ def list_windows(self) -> list[WindowInfo]:
145
+ raise PlatformError("listing windows isn't supported on wayland yet; rely on screenshots")
146
+
147
+ def focus_window(self, window_id: str) -> None:
148
+ raise PlatformError("focusing a specific window isn't supported on wayland")
149
+
150
+ def send_close_tab_shortcut(self) -> None:
151
+ if shutil.which("wtype"):
152
+ self._run("wtype", "-M", "ctrl", "w", "-m", "ctrl")
153
+ elif shutil.which("ydotool"):
154
+ # linux input event codes: 29 = left ctrl, 17 = w
155
+ self._run("ydotool", "key", "29:1", "17:1", "17:0", "29:0")
156
+ else:
157
+ raise PlatformError("sending keystrokes on wayland needs `wtype` or `ydotool`")
@@ -0,0 +1,14 @@
1
+ from focus.tools.platform.base import UnsupportedPlatform
2
+
3
+
4
+ class MacOSPlatform(UnsupportedPlatform):
5
+ """
6
+ TODO: implement on top of the shared Platform base.
7
+ - grab_screens: mss works on macOS (needs Screen Recording permission)
8
+ - list_windows / focus_window: Quartz CGWindowListCopyWindowInfo + AppleScript
9
+ - send_close_tab_shortcut: cmd+w via Quartz CGEvent
10
+ - notify: `osascript -e 'display notification ...'`
11
+ """
12
+
13
+ def __init__(self):
14
+ super().__init__("macOS")
@@ -0,0 +1,14 @@
1
+ from focus.tools.platform.base import UnsupportedPlatform
2
+
3
+
4
+ class WindowsPlatform(UnsupportedPlatform):
5
+ """
6
+ TODO: implement on top of the shared Platform base.
7
+ - grab_screens: mss works on Windows
8
+ - list_windows / focus_window: EnumWindows + SetForegroundWindow (pywin32 / ctypes)
9
+ - send_close_tab_shortcut: ctrl+w via SendInput
10
+ - notify: toast notifications (e.g. windows-toasts)
11
+ """
12
+
13
+ def __init__(self):
14
+ super().__init__("Windows")
File without changes
@@ -0,0 +1,15 @@
1
+ import typer
2
+ from rich import print
3
+
4
+
5
+ def err(message: str):
6
+ print(f"[bold red]error[/bold red]: {message}")
7
+ raise typer.Exit()
8
+
9
+
10
+ def warn(message: str):
11
+ print(f"[yellow]warning[/yellow]: {message}")
12
+
13
+
14
+ def log(message: str):
15
+ print(f"[green]log[/green]: {message}")
@@ -0,0 +1,68 @@
1
+ default_task = """
2
+
3
+ You are "Focus", the most optimal focus solution available today, designed to help your users stay in focus and complete productive work while minimizing distractions and interruptions.
4
+
5
+ Currently, your user has not specified a task to complete yet, so your job is to monitor the user and make sure they are doing "productive" work.
6
+
7
+ Examples of "productive" work include reading, writing, mathematics, homework, coding, video creation, and anything in general that "creates" and benefits your user in some way.
8
+
9
+ Examples of "unproductive" work include scrolling X, Reddit, using Discord (for unproductive tasks / checking messages / scrolling for no reason), etc.
10
+
11
+ You were created because current focus solutions are inadequate and fail to categorize "unproductive" from "productive" work accurately, instead relying on metrics like domain names. What they lack is the ability to discern whether productive work happens on unproductive apps or vice versa. Because of this, even if you suspect the user is being unproductive, wait for at minimum a couple seconds before deciding.
12
+
13
+ Your core capability is to monitor the screen -- make sure the user is being productive.
14
+
15
+ You have several tools available to you to minimize unproductive work. This includes the ability to remove tabs, notify the user through system notifications, and kill/exit apps.
16
+ """
17
+
18
+ custom_task = """
19
+
20
+ You are "Focus", the most optimal focus solution available today, designed to help your users stay in focus and complete productive work while minimizing distractions and interruptions.
21
+
22
+ For this session, your user has told you exactly what they want to get done:
23
+
24
+ <task>
25
+ {task}
26
+ </task>
27
+
28
+ Your job is to monitor the user and make sure they are working on this task. Judge activity against the task, not against generic ideas of productivity: if the task involves watching a lecture on YouTube or researching on Reddit, those are on task; reading an unrelated article or coding on a different project is not, even though it would normally look "productive". Short, task-adjacent detours (looking up documentation, checking a reference, a quick message to a collaborator about the task) are fine.
29
+
30
+ You were created because current focus solutions are inadequate and fail to categorize "unproductive" from "productive" work accurately, instead relying on metrics like domain names. What they lack is the ability to discern whether productive work happens on unproductive apps or vice versa. Because of this, even if you suspect the user is off task, wait for at minimum a couple seconds before deciding.
31
+
32
+ Your core capability is to monitor the screen -- make sure the user is working on their task.
33
+
34
+ You have several tools available to you to minimize off-task work. This includes the ability to remove tabs, notify the user through system notifications, and kill/exit apps.
35
+ """
36
+
37
+ operating_instructions = """
38
+
39
+ <how_you_operate>
40
+ You run in the background for the whole session. Every so often you get a check-in message; each check-in is a fresh look at the user's screen. A check-in already includes a screenshot, so you don't need to capture the screen before deciding whether anything looks off.
41
+
42
+ On each check-in:
43
+ 1. Look at the screenshot and decide whether the user is on task. Use list_windows when the screenshot alone is ambiguous (for example, to read a browser tab's title).
44
+ 2. If they're on task, do nothing and end the check-in.
45
+ 3. If something looks off, don't act on a single glance. Use wait for a few seconds, then capture_screen again. People glance at messages and look things up; only a sustained pattern counts as distraction.
46
+ 4. If the distraction is confirmed, intervene with the lightest action that will work, and escalate only when lighter actions have already failed in earlier check-ins:
47
+ - notify_user with a short, specific, friendly nudge.
48
+ - close_tab when a distracting tab keeps coming back after a nudge.
49
+ - kill_app only for repeated distraction in an app that has nothing to do with the task, after the user has been warned.
50
+ 5. After any close_tab or kill_app, capture_screen to confirm it did what you intended and didn't hit something the user needs.
51
+
52
+ Never close or kill anything the user plausibly needs for their task, such as their editor, terminal, documents, or reference material. When unsure, prefer a notification over a destructive action. Losing the user's work is far worse than letting a distraction run a little longer.
53
+
54
+ End every check-in with a single short line describing what you saw and what you did (for example: "On task: editing focus/agent.py in VS Code." or "YouTube gaming video for ~30s; sent a nudge."). These lines are your memory between check-ins, so make them useful to your future self.
55
+ </how_you_operate>
56
+ """
57
+
58
+ check_in = """Check-in at {time} ({elapsed} into the session).
59
+
60
+ Your notes from recent check-ins, oldest first:
61
+ {notes}
62
+
63
+ Here is the current screen."""
64
+
65
+
66
+ def build_system_prompt(task: str | None = None) -> str:
67
+ base = default_task if task is None else custom_task.format(task=task)
68
+ return base.strip() + "\n" + operating_instructions