omnipilot-agent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. omnipilot/__init__.py +19 -0
  2. omnipilot/__main__.py +6 -0
  3. omnipilot/agent/__init__.py +6 -0
  4. omnipilot/agent/demo.py +182 -0
  5. omnipilot/agent/loop.py +299 -0
  6. omnipilot/agent/prompts.py +108 -0
  7. omnipilot/agent/sessions.py +93 -0
  8. omnipilot/cli.py +1002 -0
  9. omnipilot/config/__init__.py +5 -0
  10. omnipilot/config/settings.py +408 -0
  11. omnipilot/constants.py +168 -0
  12. omnipilot/mcp/__init__.py +5 -0
  13. omnipilot/mcp/manager.py +400 -0
  14. omnipilot/permission/__init__.py +6 -0
  15. omnipilot/permission/danger.py +560 -0
  16. omnipilot/permission/policy.py +321 -0
  17. omnipilot/providers/__init__.py +22 -0
  18. omnipilot/providers/anthropic_provider.py +166 -0
  19. omnipilot/providers/base.py +127 -0
  20. omnipilot/providers/gemini_provider.py +196 -0
  21. omnipilot/providers/openai_like.py +214 -0
  22. omnipilot/providers/registry.py +124 -0
  23. omnipilot/py.typed +0 -0
  24. omnipilot/server/__init__.py +1 -0
  25. omnipilot/server/api.py +302 -0
  26. omnipilot/server/webui.py +276 -0
  27. omnipilot/tools/__init__.py +11 -0
  28. omnipilot/tools/apps.py +229 -0
  29. omnipilot/tools/base.py +170 -0
  30. omnipilot/tools/browser.py +57 -0
  31. omnipilot/tools/clipboard.py +55 -0
  32. omnipilot/tools/filesystem.py +371 -0
  33. omnipilot/tools/gui.py +227 -0
  34. omnipilot/tools/notify.py +114 -0
  35. omnipilot/tools/processes.py +171 -0
  36. omnipilot/tools/python_exec.py +116 -0
  37. omnipilot/tools/registry.py +233 -0
  38. omnipilot/tools/screen.py +166 -0
  39. omnipilot/tools/shell.py +195 -0
  40. omnipilot/tools/system_info.py +144 -0
  41. omnipilot/tools/web.py +241 -0
  42. omnipilot/ui/__init__.py +5 -0
  43. omnipilot/ui/approvals.py +131 -0
  44. omnipilot/ui/console.py +177 -0
  45. omnipilot/utils/__init__.py +1 -0
  46. omnipilot/utils/images.py +77 -0
  47. omnipilot/utils/logging_setup.py +74 -0
  48. omnipilot/utils/platform_tools.py +86 -0
  49. omnipilot/utils/strings.py +38 -0
  50. omnipilot_agent-0.1.0.dist-info/METADATA +808 -0
  51. omnipilot_agent-0.1.0.dist-info/RECORD +55 -0
  52. omnipilot_agent-0.1.0.dist-info/WHEEL +5 -0
  53. omnipilot_agent-0.1.0.dist-info/entry_points.txt +3 -0
  54. omnipilot_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
  55. omnipilot_agent-0.1.0.dist-info/top_level.txt +1 -0
omnipilot/__init__.py ADDED
@@ -0,0 +1,19 @@
1
+ """
2
+ OmniPilot — your computer, piloted by any LLM.
3
+
4
+ A cross-platform, provider-agnostic computer-use agent: it lets supported
5
+ large language models (OpenAI, Anthropic, Google Gemini, NVIDIA NIM, Groq,
6
+ OpenRouter, DeepSeek, Mistral, xAI, Together, Ollama, LM Studio and any
7
+ OpenAI-compatible endpoint) operate your machine through auditable tools —
8
+ shell commands, a Python interpreter, the filesystem, applications,
9
+ mouse/keyboard GUI control, screenshots, the clipboard, processes, web
10
+ requests and more — behind a human-in-the-loop permission system.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ __version__ = "0.1.0"
16
+ __author__ = "xrchris7"
17
+ __license__ = "MIT"
18
+
19
+ __all__ = ["__version__", "__author__", "__license__"]
omnipilot/__main__.py ADDED
@@ -0,0 +1,6 @@
1
+ """``python -m omnipilot`` entry point."""
2
+
3
+ from .cli import main_entry
4
+
5
+ if __name__ == "__main__":
6
+ main_entry()
@@ -0,0 +1,6 @@
1
+ """The agentic control loop."""
2
+
3
+ from .loop import Agent
4
+ from .sessions import SessionStore
5
+
6
+ __all__ = ["Agent", "SessionStore"]
@@ -0,0 +1,182 @@
1
+ """
2
+ Self-contained, offline demo.
3
+
4
+ A scripted provider drives the *real* agent loop against the *real* tools in a
5
+ temporary directory, so `omnipilot demo` (and the documentation asset generator)
6
+ show the full experience — plan, approval prompts, tool panels and results —
7
+ without an API key or network access.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import tempfile
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ from ..config.settings import PermissionConfig, Settings
17
+ from ..providers.base import BaseProvider, ModelResponse, ToolCall
18
+ from ..tools.base import ToolContext
19
+ from ..tools.registry import ToolRegistry
20
+ from .loop import Agent
21
+ from .prompts import build_system_prompt # noqa: F401 (re-exported for scripts)
22
+
23
+ TASK = (
24
+ "Scaffold a tiny Python CLI project called quizzer, write the code and a "
25
+ "README, run it to prove it works, then summarize what you built."
26
+ )
27
+
28
+ QUIZ_PY = '''#!/usr/bin/env python3
29
+ """Quizzer — a tiny terminal multiple-choice quiz."""
30
+
31
+ import argparse
32
+
33
+ QUESTIONS = [
34
+ {"q": "Which language is OmniPilot written in?",
35
+ "options": ["Rust", "Python", "Go"], "answer": 1},
36
+ {"q": "What does MCP stand for?",
37
+ "options": ["Model Context Protocol", "Mail Control Program",
38
+ "Minimal C Parser"], "answer": 0},
39
+ ]
40
+
41
+
42
+ def run(verbose: bool = False) -> int:
43
+ score = 0
44
+ for i, item in enumerate(QUESTIONS, start=1):
45
+ print(f"Q{i}: {item['q']}")
46
+ for j, option in enumerate(item["options"], start=1):
47
+ print(f" {j}. {option}")
48
+ if verbose:
49
+ print(f" -> answer: {item['options'][item['answer']]}")
50
+ score += 1
51
+ print(f"Score: {score}/{len(QUESTIONS)}")
52
+ return score
53
+
54
+
55
+ def main() -> None:
56
+ parser = argparse.ArgumentParser(description="Tiny terminal quiz demo")
57
+ parser.add_argument("--verbose", action="store_true", help="reveal answers")
58
+ run(verbose=parser.parse_args().verbose)
59
+
60
+
61
+ if __name__ == "__main__":
62
+ main()
63
+ '''
64
+
65
+ README_MD = """# quizzer
66
+
67
+ A tiny terminal quiz generated as an OmniPilot demo.
68
+
69
+ ## Run
70
+
71
+ ```bash
72
+ python quiz.py # take the quiz
73
+ python quiz.py --verbose # reveal answers
74
+ ```
75
+ """
76
+
77
+
78
+ class ScriptedDemoProvider(BaseProvider):
79
+ """Deterministic three-turn provider used by the offline demo."""
80
+
81
+ name = "demo"
82
+ supports_vision = False
83
+
84
+ def __init__(self) -> None:
85
+ super().__init__(model="scripted-demo-1", api_key="demo")
86
+ self._turn = 0
87
+
88
+ def complete(self, messages, tools=None, *, temperature=0.2, max_tokens=4096):
89
+ tool_results = [m for m in messages if m.get("role") == "tool"]
90
+ turn = len(tool_results)
91
+
92
+ if turn == 0:
93
+ return ModelResponse(
94
+ text=(
95
+ "I'll scaffold a small **quizzer** CLI project: create the "
96
+ "folder, write `quiz.py` and a `README.md`, then execute it to "
97
+ "verify everything works."
98
+ ),
99
+ tool_calls=[
100
+ ToolCall(id="d1", name="fs.make_directory", arguments={"path": "quizzer"}),
101
+ ToolCall(
102
+ id="d2",
103
+ name="fs.write_file",
104
+ arguments={"path": "quizzer/quiz.py", "content": QUIZ_PY},
105
+ ),
106
+ ToolCall(
107
+ id="d3",
108
+ name="fs.write_file",
109
+ arguments={"path": "quizzer/README.md", "content": README_MD},
110
+ ),
111
+ ],
112
+ )
113
+ if turn == 3:
114
+ return ModelResponse(
115
+ text=("Let's run it to verify the CLI works, including `--help`."),
116
+ tool_calls=[
117
+ ToolCall(
118
+ id="d4",
119
+ name="shell.run_command",
120
+ arguments={"command": "python quizzer/quiz.py --verbose"},
121
+ ),
122
+ ],
123
+ )
124
+ if turn == 4:
125
+ return ModelResponse(
126
+ text=(
127
+ "Done — I built **quizzer** in `./quizzer`:\n\n"
128
+ "- `quiz.py` — a 2-question terminal quiz with argparse "
129
+ "(`--verbose` reveals answers)\n"
130
+ "- `README.md` — run instructions\n\n"
131
+ "I executed `python quizzer/quiz.py --verbose` and it printed "
132
+ "both questions and scored 2/2, so the project works end to end."
133
+ ),
134
+ tool_calls=[],
135
+ )
136
+ # Defensive: no more tools.
137
+ return ModelResponse(text="(demo complete)", tool_calls=[])
138
+
139
+
140
+ def build_demo_agent(ui: Any, workdir: str | None = None) -> tuple[Agent, Any]:
141
+ """Build an Agent wired to the scripted demo provider in a temp workspace."""
142
+ import os
143
+
144
+ workspace = Path(workdir) if workdir else Path(tempfile.mkdtemp(prefix="omnipilot-demo-"))
145
+ workspace.mkdir(parents=True, exist_ok=True)
146
+ os.chdir(workspace)
147
+
148
+ settings = Settings()
149
+ settings.default_provider = "demo"
150
+ settings.permission = PermissionConfig(mode="ask", auto_approve_low=True)
151
+ settings.tools.workdir = str(workspace)
152
+
153
+ registry = ToolRegistry.build_default(settings)
154
+ provider = ScriptedDemoProvider()
155
+
156
+ from ..permission.policy import PermissionPolicy
157
+
158
+ # Approvals in the demo are auto-accepted but still rendered.
159
+ def _render_and_approve(*, tool, arguments, risk, allow_tool_wide=True):
160
+ from ..ui.approvals import render_approval
161
+
162
+ render_approval(ui.console, tool, arguments, risk, "y")
163
+ return "y"
164
+
165
+ policy = PermissionPolicy(settings, console=ui.console, prompt=_render_and_approve)
166
+
167
+ agent = Agent(
168
+ settings=settings,
169
+ provider=provider,
170
+ provider_name="demo",
171
+ registry=registry,
172
+ policy=policy,
173
+ model="offline-demo (no API key)",
174
+ ui=ui,
175
+ )
176
+ return agent, workspace
177
+
178
+
179
+ def demo_tool_context(settings: Settings, cwd: str) -> ToolContext:
180
+ import logging
181
+
182
+ return ToolContext(settings=settings, logger=logging.getLogger("demo"), console=None, cwd=cwd)
@@ -0,0 +1,299 @@
1
+ """
2
+ The agent loop: messages go in, the model replies or calls tools, calls are
3
+ permission-checked and executed, results feed back, until the model produces
4
+ a final answer (or the iteration/abort budget is exhausted).
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import logging
11
+ import os
12
+ from collections.abc import Callable
13
+ from dataclasses import dataclass, field
14
+ from pathlib import Path
15
+ from typing import Any
16
+
17
+ from ..config.settings import Settings
18
+ from ..constants import APP_NAME, MODEL_PRICES
19
+ from ..permission.policy import PermissionPolicy
20
+ from ..providers.base import (
21
+ BaseProvider,
22
+ ModelResponse,
23
+ ProviderError,
24
+ ToolCall,
25
+ Usage,
26
+ )
27
+ from ..tools.base import ToolContext, ToolResult
28
+ from ..tools.registry import ToolRegistry
29
+ from ..utils.images import EncodedImage, encode_file
30
+ from .prompts import build_system_prompt
31
+ from .sessions import SessionStore
32
+
33
+
34
+ class NullUI:
35
+ """No-op event sink (library usage / tests)."""
36
+
37
+ def assistant_message(self, text: str) -> None: ...
38
+ def tool_started(self, index: int, total: int, call: ToolCall) -> None: ...
39
+ def tool_finished(self, name: str, result: ToolResult) -> None: ...
40
+ def tool_denied(self, name: str, reason: str) -> None: ...
41
+ def info(self, message: str) -> None: ...
42
+ def error(self, message: str) -> None: ...
43
+ def system_event(self, message: str) -> None: ...
44
+
45
+
46
+ @dataclass
47
+ class Agent:
48
+ settings: Settings
49
+ provider: BaseProvider
50
+ provider_name: str
51
+ registry: ToolRegistry
52
+ policy: PermissionPolicy
53
+ model: str
54
+ ui: Any = field(default_factory=NullUI)
55
+ logger: logging.Logger | None = None
56
+ session_id: str | None = None
57
+ extra_system_prompt: str | None = None
58
+ messages: list[dict] = field(default_factory=list)
59
+ usage: Usage = field(default_factory=Usage)
60
+ cost_estimate: float = 0.0
61
+ tool_runs: int = 0
62
+ mcp_manager: Any = None
63
+
64
+ def __post_init__(self) -> None:
65
+ self.session_id = self.session_id or SessionStore().new_id()
66
+ if self.logger is None:
67
+ self.logger = logging.getLogger(APP_NAME.lower())
68
+ self.vision_enabled = self.settings.vision_enabled_for(self.provider_name)
69
+ self.context = ToolContext(
70
+ settings=self.settings,
71
+ logger=self.logger,
72
+ console=getattr(self.ui, "console", None),
73
+ cwd=self.settings.tools.workdir or os.getcwd(),
74
+ )
75
+
76
+ # -- conversation management ----------------------------------------------
77
+
78
+ def reset(self) -> None:
79
+ self.messages.clear()
80
+ self.usage = Usage()
81
+ self.cost_estimate = 0.0
82
+ self.tool_runs = 0
83
+ self.session_id = SessionStore().new_id()
84
+
85
+ def add_user_message(self, text: str, image_paths: list[str] | None = None) -> dict:
86
+ if not image_paths:
87
+ message = {"role": "user", "content": text}
88
+ else:
89
+ parts: list[dict] = [{"type": "text", "text": text or "Look at this image:"}]
90
+ for raw_path in image_paths:
91
+ path = Path(raw_path).expanduser()
92
+ encoded = _re_encode(path)
93
+ parts.append(
94
+ {
95
+ "type": "image",
96
+ "image": {"media_type": encoded.media_type, "data": encoded.data},
97
+ }
98
+ )
99
+ message = {"role": "user", "content": parts}
100
+ self.messages.append(message)
101
+ return message
102
+
103
+ def restore(self, messages: list[dict]) -> None:
104
+ """Load history from a saved session."""
105
+ self.messages = messages
106
+
107
+ # -- the loop --------------------------------------------------------------
108
+
109
+ @property
110
+ def system_prompt(self) -> str:
111
+ return build_system_prompt(
112
+ self.registry,
113
+ vision=self.vision_enabled,
114
+ permission_mode=self.settings.permission.mode,
115
+ extra=self.extra_system_prompt,
116
+ )
117
+
118
+ def run(
119
+ self,
120
+ user_text: str | None = None,
121
+ image_paths: list[str] | None = None,
122
+ should_cancel: Callable[[], bool] | None = None,
123
+ ) -> str:
124
+ if user_text:
125
+ self.add_user_message(user_text, image_paths if self.vision_enabled else None)
126
+ if image_paths and not self.vision_enabled:
127
+ self.ui.info(
128
+ "Vision is off for this provider/model; images were saved but "
129
+ "not sent to the model."
130
+ )
131
+
132
+ final_text = ""
133
+ for _step in range(1, self.settings.agent.max_iterations + 1):
134
+ if should_cancel and should_cancel():
135
+ return "(task cancelled)"
136
+ self._compact_if_needed()
137
+ request_messages = [{"role": "system", "content": self.system_prompt}, *self.messages]
138
+ try:
139
+ response: ModelResponse = self.provider.complete(
140
+ request_messages,
141
+ tools=self.registry.schemas(self.provider_name),
142
+ temperature=self.settings.agent.temperature,
143
+ max_tokens=self.settings.agent.max_tokens,
144
+ )
145
+ except ProviderError as exc:
146
+ self.ui.error(str(exc))
147
+ self.logger.error("provider error: %s", exc)
148
+ note = f"(provider error: {exc})"
149
+ self.messages.append({"role": "assistant", "content": note})
150
+ return note
151
+ except Exception as exc: # noqa: BLE001 - network oddities
152
+ self.ui.error(f"Unexpected error calling the model: {exc}")
153
+ self.logger.exception("unexpected provider failure")
154
+ return f"(unexpected error: {exc})"
155
+
156
+ self._account_usage(response)
157
+
158
+ if response.text:
159
+ self.ui.assistant_message(response.text)
160
+
161
+ if not response.tool_calls:
162
+ self.messages.append(
163
+ {
164
+ "role": "assistant",
165
+ "content": response.text or "(done)",
166
+ }
167
+ )
168
+ final_text = response.text or "(task complete)"
169
+ break
170
+
171
+ # Assistant turn requesting tools.
172
+ self.messages.append(
173
+ {
174
+ "role": "assistant",
175
+ "content": response.text or "",
176
+ "tool_calls": response.tool_calls,
177
+ }
178
+ )
179
+ if response.text:
180
+ self.ui.info(f"Planning {len(response.tool_calls)} action(s)...")
181
+
182
+ total = len(response.tool_calls)
183
+ for index, call in enumerate(response.tool_calls, 1):
184
+ if should_cancel and should_cancel():
185
+ return "(task cancelled)"
186
+ result = self._execute_tool(index, total, call)
187
+ tool_message: dict[str, Any] = {
188
+ "role": "tool",
189
+ "tool_call_id": call.id,
190
+ "name": call.name,
191
+ "content": result.message_for_model(),
192
+ "success": result.success,
193
+ }
194
+ if result.image and self.vision_enabled:
195
+ tool_message["image"] = result.image
196
+ self.messages.append(tool_message)
197
+ if not result.success and self.settings.agent.abort_on_error and not result.denied:
198
+ return f"(aborted after tool failure: {result.error})"
199
+ else:
200
+ final_text = (
201
+ f"(reached the maximum of {self.settings.agent.max_iterations} tool-calling "
202
+ "iterations; you can ask me to continue)"
203
+ )
204
+ self.ui.error(final_text)
205
+ self.messages.append({"role": "assistant", "content": final_text})
206
+
207
+ self._autosave()
208
+ return final_text
209
+
210
+ # -- tool execution --------------------------------------------------------
211
+
212
+ def _execute_tool(self, index: int, total: int, call: ToolCall) -> ToolResult:
213
+ self.ui.tool_started(index, total, call)
214
+ tool = self.registry.get(call.name)
215
+ if tool is None:
216
+ result = ToolResult.fail(
217
+ f"unknown tool {call.name!r}. Available tools: {', '.join(self.registry.names())}"
218
+ )
219
+ self.ui.tool_finished(call.name, result)
220
+ return result
221
+
222
+ arguments = call.arguments if isinstance(call.arguments, dict) else {}
223
+ try:
224
+ decision = self.policy.check(call.name, arguments, base_risk=tool.risk)
225
+ except KeyboardInterrupt: # pragma: no cover - answering the prompt
226
+ return ToolResult.denied_result("approval interrupted (Ctrl+C)")
227
+
228
+ if not decision.allowed:
229
+ result = ToolResult.denied_result(decision.reason)
230
+ self.ui.tool_denied(call.name, decision.reason)
231
+ return result
232
+
233
+ result = tool.execute(self.context, arguments)
234
+ self.ui.tool_finished(call.name, result)
235
+ self.tool_runs += 1
236
+ return result
237
+
238
+ # -- accounting / context --------------------------------------------------
239
+
240
+ def _account_usage(self, response: ModelResponse) -> None:
241
+ self.usage.add(response.usage)
242
+ for needle, (in_price, out_price) in MODEL_PRICES.items():
243
+ if needle in self.model.lower():
244
+ self.cost_estimate += (
245
+ response.usage.input_tokens / 1000 * in_price
246
+ + response.usage.output_tokens / 1000 * out_price
247
+ )
248
+ break
249
+
250
+ def _compact_if_needed(self) -> None:
251
+ """Elide verbose old tool output once the history budget is exceeded."""
252
+ budget_chars = self.settings.agent.history_token_budget * 3
253
+ total = sum(len(json.dumps(m, default=str)) for m in self.messages)
254
+ if total <= budget_chars:
255
+ return
256
+ keep_recent = 14
257
+ for msg in self.messages[:-keep_recent]:
258
+ if msg.get("role") == "tool" and len(str(msg.get("content", ""))) > 600:
259
+ msg["content"] = (
260
+ "[older tool output elided to save context; re-run the tool if needed]\n"
261
+ + str(msg["content"])[:300]
262
+ )
263
+ msg.pop("image", None)
264
+
265
+ def _autosave(self) -> None:
266
+ try:
267
+ SessionStore().save(self.session_id, self.session_payload())
268
+ except Exception as exc: # noqa: BLE001 # pragma: no cover
269
+ self.logger.debug("autosave failed: %s", exc)
270
+
271
+ def session_payload(self) -> dict[str, Any]:
272
+ return {
273
+ "id": self.session_id,
274
+ "provider": self.provider_name,
275
+ "model": self.model,
276
+ "messages": self.messages,
277
+ "usage": {
278
+ "input_tokens": self.usage.input_tokens,
279
+ "output_tokens": self.usage.output_tokens,
280
+ },
281
+ "cost_estimate": round(self.cost_estimate, 5),
282
+ }
283
+
284
+ def save(self, name: str | None = None) -> Path:
285
+ session_id = name or self.session_id
286
+ path = SessionStore().save(session_id, self.session_payload())
287
+ if name:
288
+ self.session_id = name
289
+ return path
290
+
291
+
292
+ def _re_encode(path: Path) -> EncodedImage:
293
+ image = encode_file(path)
294
+ from ..utils.images import downscale as _downscale
295
+
296
+ return EncodedImage(
297
+ media_type=image.media_type,
298
+ data=__import__("base64").b64encode(_downscale(image.raw_bytes())).decode("ascii"),
299
+ )
@@ -0,0 +1,108 @@
1
+ """System prompt construction."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import datetime
6
+ import os
7
+
8
+ from ..tools.registry import ToolRegistry
9
+ from ..utils.platform_tools import IS_WINDOWS, default_shell, platform_summary
10
+
11
+ SAFETY_RULES = """\
12
+ ## Safety and ethics (non-negotiable)
13
+ - You act on a REAL computer belonging to the user. Destructive actions (deleting files,
14
+ force-killing processes, formatting, registry changes, force pushes, privilege escalation)
15
+ trigger explicit approval. Never try to bypass, obfuscate or automate-around a denial;
16
+ the permission system is the user's, not an obstacle to defeat.
17
+ - Never read, print, upload or transmit secrets (API keys, passwords, tokens, browser
18
+ credential stores, SSH keys) unless the user explicitly asks for that exact action.
19
+ - Never run fork bombs, crypto miners, self-replicating code, ransomware-like encryption,
20
+ or anything that hides itself. Refuse and explain.
21
+ - When a command fails TWICE, stop and reason about a different approach instead of
22
+ repeating it.
23
+ - Prefer dedicated tools (fs.*, apps.*, gui.*) over shell commands when one exists;
24
+ they are safer and give structured results.
25
+ - If the user's request is ambiguous in a way that could cause damage, ask a clarifying
26
+ question (just reply with text; do not run tools speculatively).
27
+ """
28
+
29
+ GUI_RULES = """\
30
+ ## Controlling the graphical interface (computer use)
31
+ - Before ANY mouse/keyboard action, call screen.screenshot and LOOK at the returned image
32
+ to read exact coordinates and window state. Coordinates are pixels from the top-left.
33
+ - Act in small steps: screenshot -> one action -> screenshot to verify -> continue.
34
+ - Use gui.hotkey for keyboard shortcuts (ctrl+c, alt+tab, command+space, ...) and
35
+ gui.press_key for single keys. Use gui.type_text for text fields (click the field first!).
36
+ - If your click missed, re-screenshot and correct coordinates; do not spam clicks.
37
+ - Moving the mouse to the screen corner (0,0) triggers the pyautogui fail-safe abort.
38
+ """
39
+
40
+
41
+ def build_system_prompt(
42
+ registry: ToolRegistry, *, vision: bool, permission_mode: str, extra: str | None = None
43
+ ) -> str:
44
+ info = platform_summary()
45
+ shell_exe, shell_args = default_shell()
46
+ now = datetime.datetime.now().astimezone()
47
+ shell_kind = "cmd.exe (interpreter='powershell' available)" if IS_WINDOWS else shell_exe
48
+
49
+ tool_names = registry.names()
50
+ tool_catalog = "\n".join(f"- {n}" for n in tool_names)
51
+
52
+ prompt = f"""\
53
+ You are OmniPilot, an expert AI computer-use agent operating the user's real machine.
54
+ You can observe the screen, move the mouse, type, run shell commands, execute Python,
55
+ manage files, open and close applications, browse the web and automate almost anything —
56
+ through the tool functions provided to you.
57
+
58
+ ## Current environment
59
+ - Date/time: {now.strftime("%A %Y-%m-%d %H:%M %Z")}
60
+ - Operating system: {info["os_release"]}
61
+ - Host: {info["hostname"]} | logged-in user: {info["user"]}
62
+ - Current working directory: {os.getcwd()}
63
+ - Default shell: {shell_kind}
64
+ - Python: {info["python"]} (the python.run_python tool uses a persistent in-process namespace)
65
+ - Permission mode: {permission_mode} (ask = the user approves actions; auto = safe read-only
66
+ actions run unattended; yolo = no prompts)
67
+ - Vision: {
68
+ "ENABLED — you receive screenshots as images"
69
+ if vision
70
+ else "DISABLED — screenshots are only saved to disk and described to you; avoid GUI actions that require seeing the screen"
71
+ }
72
+
73
+ ## Available tools ({len(tool_names)})
74
+ {tool_catalog}
75
+
76
+ ## How you operate
77
+ 1. Understand the goal, then form a short plan. For multi-step work, briefly tell the user
78
+ the plan in one or two sentences before calling tools.
79
+ 2. Observe first: run system/list/screenshot tools to gather state; act second.
80
+ 3. Use ONE tool call per logical step when the steps depend on each other. You may batch
81
+ independent calls together.
82
+ 4. After completing the task, give a concise summary: what you did, the result, and
83
+ anything the user should check. Then STOP calling tools.
84
+ 5. If an action is denied, accept it, do not attempt workarounds, and either pick another
85
+ approach or tell the user what permission/rule is needed.
86
+ 6. Shell commands run in a fresh interpreter each time (working directory persists, but
87
+ shell variables do not). For stateful logic use python.run_python instead.
88
+ 7. Use absolute paths or paths relative to the current working directory shown above.
89
+ 8. Keep responses tight and practical; show command output only when it matters.
90
+
91
+ {
92
+ GUI_RULES
93
+ if vision
94
+ else GUI_RULES.replace(
95
+ "call screen.screenshot and LOOK at the returned image to read exact coordinates",
96
+ "vision is disabled; prefer CLI/app/API approaches and only use GUI tools when the user "
97
+ "confirms they can watch the screen",
98
+ )
99
+ }
100
+
101
+ {SAFETY_RULES}
102
+
103
+ Remember: you are granted real control. Be competent, be careful, and confirm anything
104
+ that is hard to reverse.
105
+ """
106
+ if extra:
107
+ prompt += f"\n## Additional user instructions\n{extra}\n"
108
+ return prompt