android-driver 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,144 @@
1
+ """Per-server session state: which device is selected, its driver, its last screen.
2
+
3
+ The `#N` references handed out by `screen()` are only meaningful against the
4
+ hierarchy they were produced from, so the session caches that snapshot and
5
+ invalidates it after anything that could change the display. `resolve()` refreshes
6
+ transparently when the cache is cold, which is what lets an agent call
7
+ `tap(desc=...)` without a `screen()` call in front of it.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import time
13
+ from pathlib import Path
14
+
15
+ from . import adb, ui
16
+ from .config import Config
17
+ from .drivers import Driver, create
18
+ from .log import log
19
+
20
+
21
+ class Session:
22
+ def __init__(self, cfg: Config) -> None:
23
+ self.cfg = cfg
24
+ self._serial: str | None = None
25
+ self._driver: Driver | None = None
26
+ self._elements: list[ui.Element] = []
27
+ self._elements_fresh = False
28
+ self.run_dir: Path | None = None
29
+
30
+ # ── device ───────────────────────────────────────────────────────────────
31
+
32
+ @property
33
+ def serial(self) -> str:
34
+ if self._serial is None:
35
+ self._serial = adb.pick_device()
36
+ log("session", f"auto-selected device {self._serial}")
37
+ return self._serial
38
+
39
+ @property
40
+ def current_serial(self) -> str | None:
41
+ """The pinned serial, or None — unlike `.serial`, never picks a device."""
42
+ return self._serial
43
+
44
+ def select(self, serial: str) -> None:
45
+ online = adb.list_serials()
46
+ if serial not in online:
47
+ raise ValueError(f"{serial!r} is not attached. Available: {online or '(none)'}")
48
+ self._reset()
49
+ self._serial = serial
50
+ log("session", f"device pinned to {serial}")
51
+
52
+ def reconnect(self) -> None:
53
+ """Drop the device connection so the next call rebuilds it, keeping the device pinned.
54
+
55
+ Restoring a snapshot swaps the whole OS out from under us and takes the
56
+ on-device uiautomator server with it. The connection we are holding does
57
+ not notice: it fails on the *next* request with `RemoteDisconnected`,
58
+ which reads as a broken app rather than a stale session.
59
+ """
60
+ if self._driver is not None:
61
+ try:
62
+ self._driver.close()
63
+ except Exception as e: # the far end is usually already gone
64
+ log("session", f"ignoring error while closing the driver: {e}")
65
+ self._driver = None
66
+ self.invalidate()
67
+
68
+ def _reset(self) -> None:
69
+ self.reconnect()
70
+ self._serial = None
71
+
72
+ @property
73
+ def driver(self) -> Driver:
74
+ if self._driver is None:
75
+ self._driver = create(self.serial, self.cfg.driver.backend, self.cfg.timing.click_settle_s)
76
+ return self._driver
77
+
78
+ # ── screen ───────────────────────────────────────────────────────────────
79
+
80
+ def invalidate(self) -> None:
81
+ """Mark the cached screen index stale. Call after anything that can redraw."""
82
+ self._elements_fresh = False
83
+
84
+ def refresh(self) -> list[ui.Element]:
85
+ self._elements = ui.parse(self.driver.dump_hierarchy())
86
+ self._elements_fresh = True
87
+ return self._elements
88
+
89
+ def elements(self, refresh: bool = False) -> list[ui.Element]:
90
+ if refresh or not self._elements_fresh:
91
+ return self.refresh()
92
+ return self._elements
93
+
94
+ def header(self) -> str:
95
+ app = self.driver.current_app()
96
+ width, height = self.driver.screen_size()
97
+ return (
98
+ f"device={self.serial} app={app['package'] or '?'}/{app['activity'] or '?'} "
99
+ f"screen={width}x{height} driver={self.driver.name}"
100
+ )
101
+
102
+ def resolve(self, **selector) -> ui.Element:
103
+ """Find one element, retrying once against a fresh snapshot before failing.
104
+
105
+ A stale cache is the single most common cause of a "not found" here, and
106
+ re-reading is far cheaper than making the agent debug it.
107
+ """
108
+ try:
109
+ return ui.find(self.elements(), **selector)
110
+ except LookupError:
111
+ if self._elements_fresh:
112
+ raise
113
+ return ui.find(self.refresh(), **selector)
114
+
115
+ def wait_for(self, timeout_s: float, poll_s: float = 0.5, **selector) -> ui.Element:
116
+ deadline = time.monotonic() + timeout_s
117
+ last: Exception | None = None
118
+ while True:
119
+ try:
120
+ return ui.find(self.refresh(), **selector)
121
+ except LookupError as e:
122
+ last = e
123
+ if time.monotonic() >= deadline:
124
+ raise LookupError(f"{e} (waited {timeout_s}s)") from last
125
+ time.sleep(poll_s)
126
+
127
+ def wait_until_gone(self, timeout_s: float, poll_s: float = 0.5, **selector) -> None:
128
+ """Block until nothing matches `selector`. Raises TimeoutError if it stays."""
129
+ deadline = time.monotonic() + timeout_s
130
+ while True:
131
+ try:
132
+ element = ui.find(self.refresh(), **selector)
133
+ except LookupError:
134
+ return
135
+ if time.monotonic() >= deadline:
136
+ raise TimeoutError(f"{element.label()!r} was still on screen after {timeout_s}s")
137
+ time.sleep(poll_s)
138
+
139
+ def screen_text(self) -> str:
140
+ """The compact screen index, for pasting into a failure message."""
141
+ try:
142
+ return ui.render(self.elements(), header=self.header())
143
+ except Exception as e: # a failure report must never fail
144
+ return f"(could not read the screen: {type(e).__name__}: {e})"
android_driver/ui.py ADDED
@@ -0,0 +1,261 @@
1
+ """UI hierarchy → a compact, agent-readable screen index.
2
+
3
+ The single biggest practical problem with driving Android from an LLM is that a
4
+ raw `uiautomator dump` is 50-200 KB of XML per screen. Handing that to a model
5
+ burns tens of thousands of tokens for information it mostly cannot use, and the
6
+ agent runs out of context after a handful of screens.
7
+
8
+ This module renders the same hierarchy as one line per *actionable or readable*
9
+ element, with a stable `#N` reference the agent can tap:
10
+
11
+ #1 [Button] "Sign in" desc=login_button @(540,1320)
12
+ #2 [EditText] "" desc=text_field_Email hint="Email" @(540,980)
13
+ #3 [Text] "Forgot password?" @(540,1450)
14
+
15
+ That is roughly two orders of magnitude smaller, and it reads like a menu. Raw
16
+ XML stays available through a separate, explicitly-named tool for the rare case
17
+ where the agent genuinely needs the tree.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import re
23
+ import xml.etree.ElementTree as ET
24
+ from dataclasses import dataclass
25
+
26
+ BOUNDS_RE = re.compile(r"\[(-?\d+),(-?\d+)\]\[(-?\d+),(-?\d+)\]")
27
+
28
+ # Widget classes that are always worth showing even with no text or description:
29
+ # the agent needs to know an empty input or an unlabelled toggle is there.
30
+ ALWAYS_INTERESTING = {
31
+ "android.widget.EditText",
32
+ "android.widget.Button",
33
+ "android.widget.ImageButton",
34
+ "android.widget.CheckBox",
35
+ "android.widget.RadioButton",
36
+ "android.widget.Switch",
37
+ "android.widget.SeekBar",
38
+ "android.widget.Spinner",
39
+ "android.widget.RatingBar",
40
+ }
41
+
42
+ # Friendlier names for the classes that dominate a typical dump.
43
+ CLASS_ALIASES = {
44
+ "android.widget.TextView": "Text",
45
+ "android.widget.EditText": "EditText",
46
+ "android.widget.Button": "Button",
47
+ "android.widget.ImageButton": "ImageButton",
48
+ "android.widget.ImageView": "Image",
49
+ "android.widget.CheckBox": "CheckBox",
50
+ "android.widget.RadioButton": "Radio",
51
+ "android.widget.Switch": "Switch",
52
+ "android.widget.SeekBar": "SeekBar",
53
+ "android.widget.ScrollView": "Scroll",
54
+ "android.widget.HorizontalScrollView": "HScroll",
55
+ "androidx.recyclerview.widget.RecyclerView": "List",
56
+ "android.widget.ListView": "List",
57
+ "android.view.View": "View",
58
+ "android.view.ViewGroup": "Group",
59
+ "android.widget.FrameLayout": "Frame",
60
+ "android.widget.LinearLayout": "Row",
61
+ "android.widget.RelativeLayout": "Rel",
62
+ "android.widget.Toast": "Toast",
63
+ }
64
+
65
+
66
+ @dataclass
67
+ class Element:
68
+ ref: int
69
+ cls: str
70
+ text: str
71
+ desc: str
72
+ rid: str
73
+ hint: str
74
+ bounds: tuple[int, int, int, int]
75
+ clickable: bool
76
+ scrollable: bool
77
+ checkable: bool
78
+ checked: bool
79
+ enabled: bool
80
+ focused: bool
81
+ password: bool
82
+ pkg: str
83
+
84
+ @property
85
+ def center(self) -> tuple[int, int]:
86
+ x1, y1, x2, y2 = self.bounds
87
+ return (x1 + x2) // 2, (y1 + y2) // 2
88
+
89
+ @property
90
+ def short_class(self) -> str:
91
+ return CLASS_ALIASES.get(self.cls, self.cls.rsplit(".", 1)[-1])
92
+
93
+ def label(self) -> str:
94
+ """Best human-facing name, used in error messages."""
95
+ return self.text or self.desc or self.rid.rsplit("/", 1)[-1] or self.short_class
96
+
97
+ def render(self) -> str:
98
+ parts = [f"#{self.ref}", f"[{self.short_class}]"]
99
+ if self.text or self.cls == "android.widget.EditText":
100
+ parts.append(f'"{_clip(self.text)}"')
101
+ if self.desc and self.desc != self.text:
102
+ parts.append(f"desc={_token(self.desc)}")
103
+ if self.rid:
104
+ parts.append(f"id={_token(self.rid.rsplit('/', 1)[-1])}")
105
+ if self.hint:
106
+ parts.append(f'hint="{_clip(self.hint)}"')
107
+ flags = []
108
+ if self.checkable:
109
+ flags.append("checked" if self.checked else "unchecked")
110
+ if self.scrollable:
111
+ flags.append("scrollable")
112
+ if not self.enabled:
113
+ flags.append("disabled")
114
+ if self.focused:
115
+ flags.append("focused")
116
+ if self.password:
117
+ flags.append("password")
118
+ if flags:
119
+ parts.append(f"({','.join(flags)})")
120
+ x, y = self.center
121
+ parts.append(f"@({x},{y})")
122
+ return " ".join(parts)
123
+
124
+ def to_dict(self) -> dict:
125
+ x, y = self.center
126
+ return {
127
+ "ref": f"#{self.ref}",
128
+ "class": self.short_class,
129
+ "text": self.text,
130
+ "desc": self.desc,
131
+ "id": self.rid,
132
+ "center": [x, y],
133
+ "bounds": list(self.bounds),
134
+ "clickable": self.clickable,
135
+ "enabled": self.enabled,
136
+ }
137
+
138
+
139
+ def _clip(value: str, limit: int = 80) -> str:
140
+ value = value.replace("\n", "\\n")
141
+ return value if len(value) <= limit else value[: limit - 1] + "…"
142
+
143
+
144
+ def _token(value: str) -> str:
145
+ """Quote only when the value contains whitespace — keeps common lines short."""
146
+ return value if value and not re.search(r"\s", value) else f'"{_clip(value)}"'
147
+
148
+
149
+ def _bool(node: ET.Element, attr: str) -> bool:
150
+ return node.attrib.get(attr) == "true"
151
+
152
+
153
+ def _parse_bounds(raw: str) -> tuple[int, int, int, int] | None:
154
+ m = BOUNDS_RE.match(raw or "")
155
+ if not m:
156
+ return None
157
+ x1, y1, x2, y2 = (int(g) for g in m.groups())
158
+ return x1, y1, x2, y2
159
+
160
+
161
+ def _is_interesting(node: ET.Element, bounds: tuple[int, int, int, int]) -> bool:
162
+ x1, y1, x2, y2 = bounds
163
+ if x2 <= x1 or y2 <= y1:
164
+ return False # zero-area: laid out but not visible
165
+ cls = node.attrib.get("class", "")
166
+ if cls in ALWAYS_INTERESTING:
167
+ return True
168
+ if _bool(node, "clickable") or _bool(node, "long-clickable") or _bool(node, "checkable"):
169
+ return True
170
+ if _bool(node, "scrollable"):
171
+ return True
172
+ return bool(node.attrib.get("text") or node.attrib.get("content-desc"))
173
+
174
+
175
+ def parse(xml_text: str) -> list[Element]:
176
+ """Extract the interesting elements from a hierarchy dump, in document order."""
177
+ try:
178
+ root = ET.fromstring(xml_text)
179
+ except ET.ParseError as e:
180
+ raise ValueError(f"could not parse UI hierarchy XML: {e}") from e
181
+
182
+ elements: list[Element] = []
183
+ ref = 0
184
+ for node in root.iter("node"):
185
+ bounds = _parse_bounds(node.attrib.get("bounds", ""))
186
+ if bounds is None or not _is_interesting(node, bounds):
187
+ continue
188
+ ref += 1
189
+ elements.append(
190
+ Element(
191
+ ref=ref,
192
+ cls=node.attrib.get("class", ""),
193
+ text=node.attrib.get("text", ""),
194
+ desc=node.attrib.get("content-desc", ""),
195
+ rid=node.attrib.get("resource-id", ""),
196
+ hint=node.attrib.get("hint", ""),
197
+ bounds=bounds,
198
+ clickable=_bool(node, "clickable"),
199
+ scrollable=_bool(node, "scrollable"),
200
+ checkable=_bool(node, "checkable"),
201
+ checked=_bool(node, "checked"),
202
+ enabled=_bool(node, "enabled"),
203
+ focused=_bool(node, "focused"),
204
+ password=_bool(node, "password"),
205
+ pkg=node.attrib.get("package", ""),
206
+ )
207
+ )
208
+ return elements
209
+
210
+
211
+ def render(elements: list[Element], header: str | None = None) -> str:
212
+ if not elements:
213
+ body = "(no interactive or readable elements — the screen may still be rendering)"
214
+ else:
215
+ body = "\n".join(e.render() for e in elements)
216
+ return f"{header}\n{body}" if header else body
217
+
218
+
219
+ def find(
220
+ elements: list[Element],
221
+ *,
222
+ ref: str | int | None = None,
223
+ text: str | None = None,
224
+ contains: str | None = None,
225
+ desc: str | None = None,
226
+ rid: str | None = None,
227
+ cls: str | None = None,
228
+ index: int = 0,
229
+ ) -> Element:
230
+ """Resolve a selector to exactly one element, or raise with a useful message.
231
+
232
+ Matching is deliberately strict-then-loose: `text`/`desc`/`rid` are exact so a
233
+ recipe cannot silently drift onto a different button, while `contains` exists
234
+ for the cases where the exact string is not known.
235
+ """
236
+ criteria = {"ref": ref, "text": text, "contains": contains, "desc": desc, "id": rid, "class": cls}
237
+ active = {k: v for k, v in criteria.items() if v is not None}
238
+ if not active:
239
+ raise ValueError("no selector given: pass one of ref / text / contains / desc / rid / cls")
240
+
241
+ candidates = elements
242
+ if ref is not None:
243
+ wanted = int(str(ref).lstrip("#"))
244
+ candidates = [e for e in candidates if e.ref == wanted]
245
+ if text is not None:
246
+ candidates = [e for e in candidates if e.text == text]
247
+ if contains is not None:
248
+ needle = contains.lower()
249
+ candidates = [e for e in candidates if needle in e.text.lower() or needle in e.desc.lower()]
250
+ if desc is not None:
251
+ candidates = [e for e in candidates if e.desc == desc]
252
+ if rid is not None:
253
+ candidates = [e for e in candidates if e.rid == rid or e.rid.endswith(f"/{rid}")]
254
+ if cls is not None:
255
+ candidates = [e for e in candidates if e.short_class.lower() == cls.lower() or e.cls == cls]
256
+
257
+ if not candidates:
258
+ raise LookupError(f"no element matches {active}. Call `screen` to see what is on screen.")
259
+ if index >= len(candidates):
260
+ raise LookupError(f"{active} matched {len(candidates)} element(s); index {index} is out of range")
261
+ return candidates[index]
@@ -0,0 +1,270 @@
1
+ Metadata-Version: 2.5
2
+ Name: android-driver
3
+ Version: 0.0.1
4
+ Summary: MCP server that turns an Android emulator into a deterministic, agent-drivable test harness.
5
+ Project-URL: Homepage, https://github.com/earlzdev/android-driver
6
+ Project-URL: Repository, https://github.com/earlzdev/android-driver
7
+ Project-URL: Issues, https://github.com/earlzdev/android-driver/issues
8
+ Project-URL: Documentation, https://github.com/earlzdev/android-driver#readme
9
+ Author: earlzdev
10
+ License: MIT
11
+ License-File: LICENSE
12
+ Keywords: adb,agent,android,emulator,mcp,testing,uiautomator
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Topic :: Software Development :: Quality Assurance
21
+ Classifier: Topic :: Software Development :: Testing
22
+ Requires-Python: >=3.10
23
+ Requires-Dist: mcp<2,>=1.2
24
+ Requires-Dist: pyyaml>=6.0
25
+ Requires-Dist: uiautomator2<4,>=3.2
26
+ Provides-Extra: dev
27
+ Requires-Dist: pytest>=8.0; extra == 'dev'
28
+ Requires-Dist: ruff>=0.6; extra == 'dev'
29
+ Description-Content-Type: text/markdown
30
+
31
+ # android-driver
32
+
33
+ **Drive an Android emulator as a deterministic test harness — from Claude Code.**
34
+
35
+ [![CI](https://github.com/earlzdev/android-driver/actions/workflows/ci.yml/badge.svg)](https://github.com/earlzdev/android-driver/actions/workflows/ci.yml)
36
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
37
+ [![Python 3.10+](https://img.shields.io/badge/python-3.10+-blue.svg)](https://www.python.org/downloads/)
38
+
39
+ You ask an agent to reproduce a bug in your Android app. It dumps 80 KB of accessibility XML into its
40
+ own context, taps something that turns out to be the wrong element, and when the bug does not appear
41
+ it cannot repeat what it just did — because the app is now three screens deep in a state nobody
42
+ recorded.
43
+
44
+ android-driver is built for that loop instead: **build → install → drive → assert → reset → repeat.**
45
+
46
+ ```
47
+ > reproduce the crash when the bio field goes over 100 characters
48
+
49
+ snapshot_load("clean") 1.9s
50
+ open_settings() ok
51
+ type_text(id=text_field_Bio, text=110 chars) ok
52
+ expect_log("StringIndexOutOfBoundsException") ok — matched
53
+ expect_no_crash() FAILED — 1 crash record
54
+
55
+ runs/20260901-080304-repro-set-06/report.md
56
+ → java.lang.StringIndexOutOfBoundsException: begin 0, end 120, length 110
57
+ at ...screens.SettingsScreenKt.SettingsScreen$textFields(SettingsScreen.kt:149)
58
+ ```
59
+
60
+ Three attempts from the same snapshot, three identical results, and a directory of evidence to point
61
+ at. That is the whole idea.
62
+
63
+ ---
64
+
65
+ ## Install
66
+
67
+ It is a Claude Code plugin: one install brings the tools, a skill that teaches Claude the loop, and
68
+ three slash commands.
69
+
70
+ ```bash
71
+ claude plugin marketplace add earlzdev/android-driver
72
+ claude plugin install android-driver@android-driver
73
+ ```
74
+
75
+ There is no venv to manage — the plugin builds the Python server from source on demand, and passes
76
+ your project directory to it so your config is found wherever Claude was launched from.
77
+
78
+ **Requirements:** `adb` and the Android SDK's `emulator` on `PATH`, Python ≥ 3.10. For the faster
79
+ uiautomator2 backend run `python -m uiautomator2 init` once per device; without it the server falls
80
+ back to a pure-adb backend that needs nothing installed on the device.
81
+
82
+ Prefer a plain MCP server, or not using Claude Code at all? See
83
+ [docs/installation.md](docs/installation.md).
84
+
85
+ ## Quickstart
86
+
87
+ ```
88
+ /android-driver:setup
89
+ ```
90
+
91
+ It checks your toolchain, finds your `applicationId` and build command, detects whether you are on
92
+ Compose or Views, writes a starter `.android-driver.yaml`, boots an emulator and proves the loop
93
+ works. Then:
94
+
95
+ ```
96
+ /android-driver:smoke # build, install, walk the main flows, assert nothing broke
97
+ /android-driver:repro <what is broken> # reproduce it from a snapshot and leave evidence
98
+ ```
99
+
100
+ Or just talk to it — the `android-testing` skill loads automatically when a task involves driving the
101
+ app, so "check that login still works on a fresh install" does the right thing without ceremony.
102
+
103
+ ## Why it works this way
104
+
105
+ Four decisions do most of the work.
106
+
107
+ **Snapshots, so a repro is actually reproducible.** `snapshot_save` freezes the emulator's exact
108
+ state; `snapshot_load` restores it and waits until the device is genuinely drivable again — 1.8–2.2s
109
+ measured on a Pixel 7 AVD. Reinstalling and re-navigating costs 30–90s *and drifts a little each
110
+ time*. An agent testing thirty variations of a hypothesis needs the cheap, identical option: the
111
+ difference between two attempts only means something if everything else was the same.
112
+
113
+ **A screen index instead of a wall of XML.** A raw `uiautomator dump` is 50–200 KB per screen — tens
114
+ of thousands of tokens for a model that just wants to know what it can tap. `screen` returns this:
115
+
116
+ ```
117
+ device=emulator-5554 app=com.example.app/.MainActivity screen=1080x2400 driver=uiautomator2
118
+ #5 [Scroll] id=settings_container (scrollable) @(540,1236)
119
+ #7 [Text] "Settings" id=homepage_title @(235,472)
120
+ #18 [EditText] "" id=text_field_Bio @(540,1018)
121
+ ```
122
+
123
+ Then `tap(ref="#7")`. Two orders of magnitude smaller, and it reads like a menu. The raw tree is
124
+ still there behind `dump_ui_xml` for when you genuinely need it.
125
+
126
+ **Assertions that collect their own evidence.** `expect_visible` polls, so it is safe immediately
127
+ after a tap and will not flake on an animation; when it fails it hands back the screen index that
128
+ *was* there, plus a screenshot and hierarchy dump on disk. `expect_no_crash` reads the `crash` buffer
129
+ as well as `main`, because a native abort never reaches `main` at all. Wrap a sequence in
130
+ `run_start` / `run_end` and you get `runs/<id>/` holding a timeline, a report, the logcat slice for
131
+ exactly that window, and every failure artifact — so an agent cites a directory instead of describing
132
+ what it saw.
133
+
134
+ **Your flows as first-class tools.** The six steps every test starts with — sign in, create an order,
135
+ join a call — go into `.android-driver.yaml` once and become real MCP tools with typed parameters. An
136
+ agent sees `login(email, password)` in its tool list rather than rediscovering the flow from a screen
137
+ dump every session. Recipes run the same code path as the hand-driven tools, so the two cannot drift.
138
+
139
+ <details>
140
+ <summary>Plus the device knowledge that costs an afternoon each to learn</summary>
141
+
142
+ - **Uninstall-then-install**, not `pm install -r` — debug APKs from different branches carry
143
+ different signing keys and otherwise fail with `INSTALL_FAILED_UPDATE_INCOMPATIBLE`.
144
+ - **Check `mInputShown` before pressing Back**, so dismissing the keyboard never dismisses the
145
+ dialog behind it.
146
+ - **Write to Compose `TextField`s through the accessibility node**, not tap-then-type, which lands
147
+ text in the wrong field.
148
+ - **Settle after a click** before the next query, or you read pre-animation state.
149
+ - **An `appops` pass** for OEM permission overlays that keep blocking after `pm grant` reports
150
+ success.
151
+
152
+ </details>
153
+
154
+ ## Configure
155
+
156
+ `.android-driver.yaml` at your project root is what makes a generic tool specific to your app.
157
+ `/android-driver:setup` writes a starter for you.
158
+
159
+ ```yaml
160
+ app:
161
+ package: com.example.myapp
162
+ activity: .MainActivity # optional; the launcher intent is resolved otherwise
163
+
164
+ build:
165
+ command: ./gradlew :app:assembleDebug
166
+ apk_glob: app/build/outputs/apk/debug/*.apk
167
+
168
+ driver:
169
+ backend: auto # auto | uiautomator2 | adb
170
+
171
+ selectors: # scanned, so a typo is a warning rather than a mystery
172
+ sources: ["app/src/main/**/*.kt"]
173
+
174
+ recipes: # each becomes an MCP tool with typed parameters
175
+ login:
176
+ params: {email: {required: true}, password: {required: true, secret: true}}
177
+ steps:
178
+ - launch:
179
+ - type: {desc: text_field_Email, text: "{{email}}"}
180
+ - tap: {desc: login_button}
181
+ - expect_visible: {desc: home_greeting, timeout_s: 20}
182
+ ```
183
+
184
+ The file is optional: with no config every generic tool still works — you pass `pkg=` explicitly and
185
+ lose `build_app` and recipes. It is found by walking up from your project and then, failing that, up
186
+ to three levels *down*, so an app in `app/` or `android/` is discovered without configuration.
187
+
188
+ Full reference: **[docs/configuration.md](docs/configuration.md)** · recipe and step syntax:
189
+ **[docs/recipes.md](docs/recipes.md)** · worked examples for Compose and View projects:
190
+ [`examples/`](examples/).
191
+
192
+ ## Tools
193
+
194
+ 45, plus one per configured recipe.
195
+
196
+ | Group | Tools |
197
+ |---|---|
198
+ | Emulator | `list_avds` `start_emulator` `stop_emulator` `wait_for_boot` `snapshot_save` `snapshot_load` `snapshot_list` `snapshot_delete` |
199
+ | Device | `list_devices` `select_device` `device_info` |
200
+ | App | `build_app` `install_app` `uninstall_app` `app_info` `launch_app` `force_stop` `clear_app_data` |
201
+ | UI | `screen` `tap` `tap_xy` `long_press` `type_text` `swipe` `scroll_to` `press_key` `screenshot` `dump_ui_xml` |
202
+ | Assertions | `expect_visible` `expect_gone` `expect_log` `expect_no_crash` |
203
+ | Runs | `run_start` `run_end` `run_list` `record_start` `record_stop` |
204
+ | Recipes | `list_recipes` `run_recipe` `check_recipes` `list_selectors` `reload_config` |
205
+ | Logs | `logcat_clear` `logcat_read` |
206
+ | Shell | `shell` |
207
+
208
+ > [!WARNING]
209
+ > `shell` is unrestricted on purpose — this is a development tool, not a sandbox. It can wipe device
210
+ > data, kill processes and read files. Point it at emulators and test devices, not at anything you
211
+ > care about.
212
+
213
+ ### The loop, in tool calls
214
+
215
+ ```python
216
+ start_emulator(avd="Pixel_7") # reuses one that is already running
217
+ install_app(build_first=True)
218
+ launch_app()
219
+ snapshot_save("clean") # ← the state every attempt returns to
220
+
221
+ run_start("issue 412: crash on empty search")
222
+ … login() / tap / type_text / expect_visible …
223
+ expect_no_crash()
224
+ run_end() # → runs/<id>/report.md
225
+
226
+ snapshot_load("clean") # next variation, from identical state
227
+ ```
228
+
229
+ [`docs/agent-guide.md`](docs/agent-guide.md) is a `CLAUDE.md` fragment you can drop into a project so
230
+ an agent picks this up without being told.
231
+
232
+ ## Try it without your own app
233
+
234
+ The repo ships **[FlakyDemo](test_app/)** — a Compose app with five screens and **27 deliberately
235
+ planted bugs**, each documented in [`test_app/BUGS.md`](test_app/BUGS.md) with what was actually
236
+ observed rather than what was intended. Crashes, races, state lost on rotation, a Save button that
237
+ reports success and silently does nothing, and several bugs that only show up one run in three.
238
+
239
+ Its flake generator is seeded, so `--el flake_seed 42` replays the same failures every time and
240
+ `--ez flake_enabled false` turns them all off for a clean baseline. It ships with 11 recipes.
241
+
242
+ It is the fastest way to see what the tool is for — and a fair test of whether it earns its keep,
243
+ since a good number of the planted bugs are the kind a tap-and-screenshot agent cannot catch at all.
244
+ The Settings screen alone reports "Saved" in the UI while the log says
245
+ `outcome=noop reason=terms_not_accepted` and nothing was written.
246
+
247
+ ## Development
248
+
249
+ ```bash
250
+ uv sync --extra dev
251
+ uv run pytest # 129 unit tests against a fake driver, no device needed
252
+ uv run ruff check src tests
253
+ ```
254
+
255
+ Setup, the live suite, and the packaging pitfalls worth knowing about are in
256
+ **[CONTRIBUTING.md](CONTRIBUTING.md)**.
257
+
258
+ ## Status
259
+
260
+ Early, but working end to end. Emulator lifecycle, snapshots, both driver backends, the screen index,
261
+ assertions, run bundles, screen recording, recipes and selector scanning are implemented and covered
262
+ by tests; the live suite passes against a Pixel 7 AVD. Packaged as a Claude Code plugin with a skill
263
+ and three commands.
264
+
265
+ Not on PyPI — the plugin builds from source, so it does not need to be. Roadmap:
266
+ [docs/roadmap.md](docs/roadmap.md). Issues and pull requests welcome.
267
+
268
+ ## License
269
+
270
+ MIT