ultracua 0.182.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ultracua-0.182.0/LICENSE +21 -0
- ultracua-0.182.0/PKG-INFO +214 -0
- ultracua-0.182.0/README.md +182 -0
- ultracua-0.182.0/pyproject.toml +106 -0
- ultracua-0.182.0/src/ultracua/__init__.py +103 -0
- ultracua-0.182.0/src/ultracua/agent.py +88 -0
- ultracua-0.182.0/src/ultracua/audit.py +535 -0
- ultracua-0.182.0/src/ultracua/browser.py +524 -0
- ultracua-0.182.0/src/ultracua/cache.py +428 -0
- ultracua-0.182.0/src/ultracua/cli.py +1413 -0
- ultracua-0.182.0/src/ultracua/conditions.py +44 -0
- ultracua-0.182.0/src/ultracua/config.py +176 -0
- ultracua-0.182.0/src/ultracua/contracts.py +352 -0
- ultracua-0.182.0/src/ultracua/daemon/__init__.py +12 -0
- ultracua-0.182.0/src/ultracua/daemon/__main__.py +8 -0
- ultracua-0.182.0/src/ultracua/daemon/client.py +51 -0
- ultracua-0.182.0/src/ultracua/daemon/server.py +150 -0
- ultracua-0.182.0/src/ultracua/dryrun.py +364 -0
- ultracua-0.182.0/src/ultracua/extract.py +111 -0
- ultracua-0.182.0/src/ultracua/flow.py +2013 -0
- ultracua-0.182.0/src/ultracua/flows.py +5435 -0
- ultracua-0.182.0/src/ultracua/fsio.py +97 -0
- ultracua-0.182.0/src/ultracua/history.py +111 -0
- ultracua-0.182.0/src/ultracua/ledger.py +147 -0
- ultracua-0.182.0/src/ultracua/llm/__init__.py +60 -0
- ultracua-0.182.0/src/ultracua/llm/anthropic.py +122 -0
- ultracua-0.182.0/src/ultracua/llm/base.py +87 -0
- ultracua-0.182.0/src/ultracua/llm/gemini.py +114 -0
- ultracua-0.182.0/src/ultracua/llm/mock.py +28 -0
- ultracua-0.182.0/src/ultracua/llm/openai.py +123 -0
- ultracua-0.182.0/src/ultracua/llm/types.py +95 -0
- ultracua-0.182.0/src/ultracua/locators.py +779 -0
- ultracua-0.182.0/src/ultracua/mcpserver/__init__.py +31 -0
- ultracua-0.182.0/src/ultracua/mcpserver/server.py +568 -0
- ultracua-0.182.0/src/ultracua/obs.py +379 -0
- ultracua-0.182.0/src/ultracua/parallel.py +58 -0
- ultracua-0.182.0/src/ultracua/pin.py +116 -0
- ultracua-0.182.0/src/ultracua/providers/__init__.py +51 -0
- ultracua-0.182.0/src/ultracua/providers/base.py +72 -0
- ultracua-0.182.0/src/ultracua/providers/llm_agent.py +149 -0
- ultracua-0.182.0/src/ultracua/providers/mock.py +47 -0
- ultracua-0.182.0/src/ultracua/providers/scripted.py +60 -0
- ultracua-0.182.0/src/ultracua/recorder.py +893 -0
- ultracua-0.182.0/src/ultracua/safety.py +398 -0
- ultracua-0.182.0/src/ultracua/snapshot.py +491 -0
- ultracua-0.182.0/src/ultracua/timing.py +45 -0
- ultracua-0.182.0/src/ultracua/types.py +97 -0
- ultracua-0.182.0/src/ultracua/verifiers.py +56 -0
- ultracua-0.182.0/src/ultracua/verify.py +15 -0
- ultracua-0.182.0/src/ultracua/vision.py +131 -0
- ultracua-0.182.0/src/ultracua/webmcp.py +55 -0
ultracua-0.182.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Raimondas Lencevicius
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ultracua
|
|
3
|
+
Version: 0.182.0
|
|
4
|
+
Summary: A Computer Use Agent that learns a browser flow once with an LLM, then replays it deterministically with no planning model in the loop.
|
|
5
|
+
Keywords: browser-automation,computer-use-agent,playwright,llm-agent,deterministic-replay,web-scraping,rpa
|
|
6
|
+
Author: Raimondas Lencevicius
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Classifier: Development Status :: 7 - Inactive
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Browsers
|
|
15
|
+
Classifier: Topic :: Software Development :: Testing
|
|
16
|
+
Requires-Dist: anthropic>=0.109.2,<1.0
|
|
17
|
+
Requires-Dist: playwright>=1.60.0,<2.0
|
|
18
|
+
Requires-Dist: pydantic>=2.13.4,<3.0
|
|
19
|
+
Requires-Dist: python-dotenv>=1.2.2,<2.0
|
|
20
|
+
Requires-Dist: xxhash>=3.7.0,<5.0
|
|
21
|
+
Requires-Dist: mcp>=1.28.0,<2.0 ; extra == 'mcp'
|
|
22
|
+
Requires-Dist: google-genai>=2.8.0,<3.0 ; extra == 'providers'
|
|
23
|
+
Requires-Dist: openai>=2.42.0,<3.0 ; extra == 'providers'
|
|
24
|
+
Requires-Python: >=3.12
|
|
25
|
+
Project-URL: Repository, https://github.com/raimondasl/ultracua
|
|
26
|
+
Project-URL: Defect register, https://github.com/raimondasl/ultracua/blob/main/docs/open-defects.md
|
|
27
|
+
Project-URL: Honest status, https://github.com/raimondasl/ultracua/blob/main/STATUS.md
|
|
28
|
+
Project-URL: Benchmark numbers, https://github.com/raimondasl/ultracua/blob/main/baselines/README.md
|
|
29
|
+
Provides-Extra: mcp
|
|
30
|
+
Provides-Extra: providers
|
|
31
|
+
Description-Content-Type: text/markdown
|
|
32
|
+
|
|
33
|
+
# ultracua
|
|
34
|
+
|
|
35
|
+
[](https://github.com/raimondasl/ultracua/actions/workflows/ci.yml)
|
|
36
|
+
|
|
37
|
+
A Computer Use Agent (CUA) that **learns a browser flow once with an LLM, then replays it
|
|
38
|
+
deterministically** — not by clicking faster, but by taking the planning model out of the loop.
|
|
39
|
+
A replayed **write** makes no model call at all **unless it extracts a readback** (a confirmation
|
|
40
|
+
number, say) — measured: 0 calls on all 24 write replays in the committed series, of which 4
|
|
41
|
+
completed the write, 9 put idempotency-keyed requests on the wire, and 15 were refused before acting
|
|
42
|
+
by open gate defects (`R4.111`, `R4.148`). A **data read** makes one call, to read the answer off the
|
|
43
|
+
page: one on 43 of 51 read replays, two on the other 8.
|
|
44
|
+
A write flow that sets `extract=` takes that same one call — `flows.py` builds an extraction router
|
|
45
|
+
whenever `spec.extract` is set and the read is not pinned, and it does not ask whether the flow
|
|
46
|
+
writes. That holds for `replay_flow()` / `flow replay`, which run
|
|
47
|
+
`mode="replay"`; the one-shot CLI (`ultracua --url URL --goal GOAL`) defaults to `--mode auto`, which
|
|
48
|
+
may self-heal or re-author and is *not* 0-LLM.
|
|
49
|
+
|
|
50
|
+
> **Status: active development stopped on 2026-09-09.** (Releases after that date are close-out
|
|
51
|
+
> work only — making the record honest, freezing the programme, and publishing once. This is
|
|
52
|
+
> 0.182.0, the release that publishes.) This is a finished
|
|
53
|
+
> experiment, not a maintained product. Everything described below was measured and the measurements
|
|
54
|
+
> stand; what stopped is new work. There are **75 open defects**, published in
|
|
55
|
+
> [docs/open-defects.md](https://github.com/raimondasl/ultracua/blob/main/docs/open-defects.md) along with what each one costs you — reading that
|
|
56
|
+
> register is the honest way to decide whether this fits your case. The benchmark numbers quoted here
|
|
57
|
+
> are dated observations that will not be re-cut, and the baseline files themselves record no version
|
|
58
|
+
> or date, so the dates given here and in `baselines/README.md` are all the provenance they have.
|
|
59
|
+
> The method this project was really about — how a solo codebase keeps itself honest when its own
|
|
60
|
+
> test suite is not the instrument that finds its bugs — is written up in `CLAUDE.md` and the
|
|
61
|
+
> register, and both are worth more than the code.
|
|
62
|
+
|
|
63
|
+
It sits between two unsatisfying options:
|
|
64
|
+
|
|
65
|
+
- **Hand-coded scripts** (Playwright / Selenium) — fast and free, but a developer hand-writes every
|
|
66
|
+
selector and they break on UI changes.
|
|
67
|
+
- **Per-step LLM agents** (browser-use, computer-use) — handle novelty, but run the model *every
|
|
68
|
+
step, every time*: slow, expensive, non-deterministic, hard to audit.
|
|
69
|
+
|
|
70
|
+
ultracua's middle path: an LLM **authors the flow once** (no hand-scripting), it **self-heals** minor
|
|
71
|
+
UI drift, then **replays the navigation with zero LLM** — fast, cheap, deterministic, auditable.
|
|
72
|
+
(A data read still pays one extraction call; only writes and navigate-only reads are 0-LLM end to end.)
|
|
73
|
+
|
|
74
|
+
## What it's for, and what it never became
|
|
75
|
+
|
|
76
|
+
**Good fit** — *repeated* browser automation: scheduled data pulls from authenticated dashboards,
|
|
77
|
+
internal tooling, portal extractions where the flow is stable and run often. *"Every morning, log
|
|
78
|
+
into the vendor portal and pull yesterday's order count."* Learning a flow costs a handful of model
|
|
79
|
+
calls and cents — median **4 calls** and **$0.07** over the 87 measured learns in the committed
|
|
80
|
+
benchmark series (range 2–13 calls, $0.03–$0.18). Each later run re-drives the recipe with no
|
|
81
|
+
planning model in the loop. For an end-to-end wall-clock figure, run the shipped example: on
|
|
82
|
+
2026-09-09 `examples/hn_digest.py` learned in 30 s and replayed in 5 s against a live site — one
|
|
83
|
+
flow, one run, so treat it as an illustration rather than a benchmark.
|
|
84
|
+
|
|
85
|
+
**Not a fit** — a no-code "do anything" agent; one-off complex analysis (use a per-step LLM agent
|
|
86
|
+
for those); anything high-stakes run unsupervised without a human verifying the learned flow. These
|
|
87
|
+
said *not yet* until 0.181.0; development stopped, so they are simply not a fit.
|
|
88
|
+
|
|
89
|
+
### What this is now — the scope the measurements actually support
|
|
90
|
+
|
|
91
|
+
Read this before adopting it. The claims above hold **inside a scope that has been measured on two
|
|
92
|
+
real applications**, and the honest boundary is narrower than the feature list below:
|
|
93
|
+
|
|
94
|
+
- **Supervised authoring, unattended replay.** A human reviews the learned recipe and approves it;
|
|
95
|
+
the approval is bound to a digest of the steps reviewed, so a re-authored recipe refuses rather
|
|
96
|
+
than running. Reads then run unattended. A **declared** write (`spec.mutate` set) sits behind a
|
|
97
|
+
human checkpoint — `flow dry-run` → `flow inspect` → `flow approve`. A write the classifier merely
|
|
98
|
+
*infers* is gated, idempotency-keyed and refused a re-author, but is **not** approval-gated.
|
|
99
|
+
- **Measured on:** a server-rendered app (Gitea 1.22) and one client-rendered SPA (Odoo 17 CRM),
|
|
100
|
+
seven scenarios each, three passes each. `availability_rate` **0.762** (Gitea, cut 2026-08-26) and
|
|
101
|
+
**0.714** (Odoo, cut 2026-09-05 from a series run at 0.169.0), each over n=21 scenario-observations.
|
|
102
|
+
**Neither has been re-cut since**, and neither file records its own cut date or version. A later
|
|
103
|
+
Gitea series measured 0.857 and was **never cut as a baseline**, so that figure has no committed
|
|
104
|
+
artifact behind it — read all three as a level, not a current rate. No silently-wrong outcome is recorded in any committed
|
|
105
|
+
customer-benchmark series (`inviolable: []` throughout); note that `drift_bench`, a different
|
|
106
|
+
instrument, publishes its own non-zero wrong-bind allowlist. The rows that do not pass are named
|
|
107
|
+
and diagnosed in [baselines/README.md](https://github.com/raimondasl/ultracua/blob/main/baselines/README.md).
|
|
108
|
+
- **Expect first-run engineering on a new application.** Reaching that number on the SPA took ten
|
|
109
|
+
`src/` fixes across ~40 releases (0.133.0 → 0.172.0): render readiness, reads served over POST,
|
|
110
|
+
off-screen controls, an element cap, a scroll that landed on the wrong container. Most are general
|
|
111
|
+
and now help both apps — but a third kind of app should be assumed to need its own.
|
|
112
|
+
- **Fails loud, by design:** anti-bot / CAPTCHA interstitials (the run escalates rather than burning
|
|
113
|
+
retries), an ambiguous locator match, a write whose form scope drifted (never LLM-healed), a login
|
|
114
|
+
the built-in form-filler cannot complete, and a WebSocket frame seen during `flow record` or
|
|
115
|
+
`flow dry-run`.
|
|
116
|
+
- **Not supported — and these are gaps, not refusals.** 2FA / SSO logins need a user-supplied
|
|
117
|
+
callable and are not first-class. **iframe and shadow-DOM** content is not captured at all (top
|
|
118
|
+
frame only), and a sub-frame write is deliberately excluded from write reconciliation, so an iframe
|
|
119
|
+
write triggered by a recorded action can cache **ungated**. A write issued from a **shared worker**
|
|
120
|
+
is invisible to every request watcher Playwright offers (`R4.154`, open) — it can double-submit at
|
|
121
|
+
learn and cache as a read. Reads served over **GraphQL POST** are still classified as writes
|
|
122
|
+
(`R4.27`, open); through `replay()` that is a loud refusal, but through a `mode="auto"` door it
|
|
123
|
+
quietly falls back to a re-author and you lose the 0-LLM replay.
|
|
124
|
+
|
|
125
|
+
It's a **usable prototype of a real pattern, not a turnkey product.** The honest status, the measured
|
|
126
|
+
benchmark numbers, and the known gaps live in **[STATUS.md](https://github.com/raimondasl/ultracua/blob/main/STATUS.md)**.
|
|
127
|
+
|
|
128
|
+
## Install
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
pip install ultracua
|
|
132
|
+
python -m playwright install chromium # one-time browser download; nothing works without it
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
Optional extras, for the surfaces the core does not pull:
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
pip install "ultracua[providers]" # the OpenAI and Gemini adapters
|
|
139
|
+
pip install "ultracua[mcp]" # `ultracua flow serve-mcp`
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
Python **3.12** — the only version this has ever been tested on. The dependency floors are capped at
|
|
143
|
+
the next major on purpose: nothing will update this package again, and an uncapped floor is a
|
|
144
|
+
guarantee of eventual breakage rather than of future compatibility.
|
|
145
|
+
|
|
146
|
+
### Working on the source instead
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
uv sync --all-groups # create the venv + install deps (uv manages Python too)
|
|
150
|
+
# NOT bare `uv sync` — it strips the bench/providers/mcp
|
|
151
|
+
# groups and the test suite then fails to import
|
|
152
|
+
uv run playwright install chromium # one-time browser download
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
You'll need `ANTHROPIC_API_KEY` (e.g. in a gitignored `.env`) for the one-time *learn* run and the
|
|
156
|
+
per-run extraction call.
|
|
157
|
+
|
|
158
|
+
## Quickstart
|
|
159
|
+
|
|
160
|
+
Define a recurring task once, learn it, then replay it with 0-LLM navigation — it returns structured data and
|
|
161
|
+
**fails loud** on drift instead of returning a wrong value:
|
|
162
|
+
|
|
163
|
+
```python
|
|
164
|
+
import asyncio
|
|
165
|
+
from ultracua import FlowSpec, learn_flow, approve_flow, replay_flow
|
|
166
|
+
|
|
167
|
+
spec = FlowSpec(
|
|
168
|
+
name="daily-orders",
|
|
169
|
+
start_url="https://portal.example.com/admin",
|
|
170
|
+
goal="open the orders report",
|
|
171
|
+
extract="the number of orders placed yesterday", # → structured data
|
|
172
|
+
)
|
|
173
|
+
res = asyncio.run(learn_flow(spec)) # author once; inspect res.steps / res.data
|
|
174
|
+
approve_flow(spec) # a human verifies before trusting it unattended
|
|
175
|
+
data = asyncio.run(replay_flow(spec)) # every run after: 0-LLM navigation, returns the data
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
Prefer a real, runnable walkthrough? **[EXAMPLES.md](https://github.com/raimondasl/ultracua/blob/main/EXAMPLES.md)** does this end-to-end against
|
|
179
|
+
Hacker News (read-only) and is built to record: `uv run python examples/hn_digest.py --headed`.
|
|
180
|
+
|
|
181
|
+
## Highlights
|
|
182
|
+
|
|
183
|
+
- **0-LLM navigation** — through `replay_flow()` / `flow replay` (`mode="replay"`) a learned flow re-drives every step with no model call; this is pinned key-lessly by `tests/test_inviolable_properties.py`, which sweeps the drift corpus with a floor of 90 rows (99 on the last run) and asserts per row that replay resolves no provider at all. A write with no readback is then 0-LLM end to end (0 calls on all 24 write replays in the committed series — see the headline for what those 24 contain); a *data* read, **and any write that sets `extract=`**, adds one extraction call — measured **1 call on 43 of 51 read replays, 2 on the other 8**, where an auth-refresh retry re-extracts. `pin_read` would remove even that, but it only fires for a scalar answer that maps to exactly one element carrying an `id`/`data-testid`: measured pinnable on **0 of the corpus's 10 read scenarios**, so budget one call per read.
|
|
184
|
+
- **Resilient, self-healing locators** — survive cosmetic DOM drift; one-step LLM re-grounding on real drift, or a **suffix-replan** that re-authors just the broken tail (keeping the working prefix) when the path changes. **Measured**, key-lessly and in CI on a 187-row corpus: 0-LLM survival degrades with mutation intensity and reaches zero — 20/27 at k=1, 0/6 at k=7, non-monotonic in between (k50 = 6) — and the heal machinery recovers **36 of 39** heal-eligible rows *given* correct element identity ([drift-bench v2](https://github.com/raimondasl/ultracua/blob/main/baselines/README.md) — read the limits; that heal figure is a mechanism ceiling measured against a perfect-vision oracle, not a heal rate; the committed `drift_v2.json` was two releases stale when this line first said so and was re-recorded at 0.181.0, R4.159). Two wrong-bind classes are **published rather than hidden**, each pinned by its own corpus row.
|
|
185
|
+
- **Trust controls** — an approval gate **bound to the steps you reviewed** (re-author them and replay refuses, before the browser opens), data-shape + value-contract drift detection, **fail-loud** `FlowReplayError`.
|
|
186
|
+
- **Auth refresh** — re-login on session expiry; credentials are env-sourced and **never persisted**.
|
|
187
|
+
- **Write flows** — submit / post / purchase with **action-completion verification** + idempotency.
|
|
188
|
+
- **Dry run** — `flow dry-run` replays a write flow with **every write held**: see the exact body, URL and idempotency key it *would* send before you approve it. Anything that cannot be *proven* held aborts rather than proceeding — **within the realms a watcher can see**. A write issued from a **shared worker** is invisible to every request watcher Playwright offers (`R4.154`, open, measured on both benchmark substrates), so for that realm neither the hold nor the abort can fire.
|
|
189
|
+
- **Record by demonstration** — `ultracua flow record` captures a headed walkthrough into a cached **0-LLM** flow: reads are verify-by-replay; declared writes are **gated + approval-gated + idempotency-keyed**.
|
|
190
|
+
- **Fleet supervisor** — `flow run-all` replays every saved flow, reports pass/fail, alerts, exits non-zero for cron — including when a flow was refused a run by something no human chose, or when nothing ran at all; `flow status` for history.
|
|
191
|
+
- **Multi-provider** — Anthropic / OpenAI / Gemini, fast/strong tiering, prompt caching.
|
|
192
|
+
- **Drive from any language** — JSON-RPC daemon + a Node/JS client.
|
|
193
|
+
|
|
194
|
+
## Documentation
|
|
195
|
+
|
|
196
|
+
| Doc | For |
|
|
197
|
+
|---|---|
|
|
198
|
+
| **[EXAMPLES.md](https://github.com/raimondasl/ultracua/blob/main/EXAMPLES.md)** | a worked, runnable real-site example — **start here** |
|
|
199
|
+
| **[GUIDE.md](https://github.com/raimondasl/ultracua/blob/main/GUIDE.md)** | developer guide: the Flow API + CLI in depth (auth, write flows, record by demonstration, health, providers) |
|
|
200
|
+
| **[HEALING.md](https://github.com/raimondasl/ultracua/blob/main/HEALING.md)** | how it self-heals (and deliberately doesn't) when a page's elements change: resilient locators, LLM heal/re-plan, and the fail-loud boundaries |
|
|
201
|
+
| **[docs/comparison-stagehand.md](https://github.com/raimondasl/ultracua/blob/main/docs/comparison-stagehand.md)** | ultracua vs. Stagehand — a design-philosophy comparison (drift/self-heal, write safety, data correctness), dated + sourced |
|
|
202
|
+
| **[docs/open-defects.md](https://github.com/raimondasl/ultracua/blob/main/docs/open-defects.md)** | the standing defect register — **four** adversarial rounds. Rounds 1–2 are fixed; of the 13 R3-numbered entries (round 3 found 11, R3.12–R3.13 were filed later), **2 remain open** (R3.2, R3.7); round 4 began as a pre-merge audit that *parked* a change rather than ship it and became a standing per-slice log — **160 findings, 73 open / 83 fixed / 4 parked**. Records the residuals it decided not to fix, including the decisions closed as NO CHANGE. Its R4 index is rendered from `docs/register/state.json` and its R3 count is parsed from the headings — both machine-checked; **this table row is prose and is not**, which is why it said "nine remain open" for five weeks after that stopped being true. Check the register's own index, not this row. **Read before starting new work.** |
|
|
203
|
+
| **[docs/correctness-plan.md](https://github.com/raimondasl/ultracua/blob/main/docs/correctness-plan.md)** | **closed at 0.181.0** — the plan behind Phases 0–7, grounded on the v0.75.0 survey (round-3 findings, test-machinery holes, unpinned residuals) — worst user harm first, test-first, one slice per PR. It does **not** sequence the round-4 register: most of the 73 open findings post-date it, and `docs/reshape-plan.md` §13 plus `docs/plan/state.json` are the operative order. |
|
|
204
|
+
| **[docs/correctness-survey.md](https://github.com/raimondasl/ultracua/blob/main/docs/correctness-survey.md)** | the measured inventory the plan is built on: **59 items** across the register, CI/eval machinery, the user-facing surface and the accepted residuals in `src/`, produced 2026-08-04 at v0.75.0 and not re-derived since. |
|
|
205
|
+
| **[ARCHITECTURE.md](https://github.com/raimondasl/ultracua/blob/main/ARCHITECTURE.md)** | how it works inside + how to contribute (engine, safety, tiers, benchmarks, layout) |
|
|
206
|
+
| **[STATUS.md](https://github.com/raimondasl/ultracua/blob/main/STATUS.md)** | honest status, measured benchmarks, known fragilities |
|
|
207
|
+
| **[ROADMAP.md](https://github.com/raimondasl/ultracua/blob/main/ROADMAP.md)** | what was *going* to be next — **closed at 0.181.0**, kept as the record of what was intended. Its opening "2–7×" claim is unsupported and is corrected in its own banner |
|
|
208
|
+
| **[evals/README.md](https://github.com/raimondasl/ultracua/blob/main/evals/README.md)** | the manual capability-eval suite: shipped behavior + the H1–H16 horizons, with $ cost estimates and partial runs |
|
|
209
|
+
| **[PLAN.md](https://github.com/raimondasl/ultracua/blob/main/PLAN.md)** | the original design + research basis |
|
|
210
|
+
| **[docs/recorder-spike.md](https://github.com/raimondasl/ultracua/blob/main/docs/recorder-spike.md)** | the record-by-demonstration design (capture → gated cache → 0-LLM replay) |
|
|
211
|
+
|
|
212
|
+
## License
|
|
213
|
+
|
|
214
|
+
[MIT](https://github.com/raimondasl/ultracua/blob/main/LICENSE) © Raimondas Lencevicius
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
# ultracua
|
|
2
|
+
|
|
3
|
+
[](https://github.com/raimondasl/ultracua/actions/workflows/ci.yml)
|
|
4
|
+
|
|
5
|
+
A Computer Use Agent (CUA) that **learns a browser flow once with an LLM, then replays it
|
|
6
|
+
deterministically** — not by clicking faster, but by taking the planning model out of the loop.
|
|
7
|
+
A replayed **write** makes no model call at all **unless it extracts a readback** (a confirmation
|
|
8
|
+
number, say) — measured: 0 calls on all 24 write replays in the committed series, of which 4
|
|
9
|
+
completed the write, 9 put idempotency-keyed requests on the wire, and 15 were refused before acting
|
|
10
|
+
by open gate defects (`R4.111`, `R4.148`). A **data read** makes one call, to read the answer off the
|
|
11
|
+
page: one on 43 of 51 read replays, two on the other 8.
|
|
12
|
+
A write flow that sets `extract=` takes that same one call — `flows.py` builds an extraction router
|
|
13
|
+
whenever `spec.extract` is set and the read is not pinned, and it does not ask whether the flow
|
|
14
|
+
writes. That holds for `replay_flow()` / `flow replay`, which run
|
|
15
|
+
`mode="replay"`; the one-shot CLI (`ultracua --url URL --goal GOAL`) defaults to `--mode auto`, which
|
|
16
|
+
may self-heal or re-author and is *not* 0-LLM.
|
|
17
|
+
|
|
18
|
+
> **Status: active development stopped on 2026-09-09.** (Releases after that date are close-out
|
|
19
|
+
> work only — making the record honest, freezing the programme, and publishing once. This is
|
|
20
|
+
> 0.182.0, the release that publishes.) This is a finished
|
|
21
|
+
> experiment, not a maintained product. Everything described below was measured and the measurements
|
|
22
|
+
> stand; what stopped is new work. There are **75 open defects**, published in
|
|
23
|
+
> [docs/open-defects.md](https://github.com/raimondasl/ultracua/blob/main/docs/open-defects.md) along with what each one costs you — reading that
|
|
24
|
+
> register is the honest way to decide whether this fits your case. The benchmark numbers quoted here
|
|
25
|
+
> are dated observations that will not be re-cut, and the baseline files themselves record no version
|
|
26
|
+
> or date, so the dates given here and in `baselines/README.md` are all the provenance they have.
|
|
27
|
+
> The method this project was really about — how a solo codebase keeps itself honest when its own
|
|
28
|
+
> test suite is not the instrument that finds its bugs — is written up in `CLAUDE.md` and the
|
|
29
|
+
> register, and both are worth more than the code.
|
|
30
|
+
|
|
31
|
+
It sits between two unsatisfying options:
|
|
32
|
+
|
|
33
|
+
- **Hand-coded scripts** (Playwright / Selenium) — fast and free, but a developer hand-writes every
|
|
34
|
+
selector and they break on UI changes.
|
|
35
|
+
- **Per-step LLM agents** (browser-use, computer-use) — handle novelty, but run the model *every
|
|
36
|
+
step, every time*: slow, expensive, non-deterministic, hard to audit.
|
|
37
|
+
|
|
38
|
+
ultracua's middle path: an LLM **authors the flow once** (no hand-scripting), it **self-heals** minor
|
|
39
|
+
UI drift, then **replays the navigation with zero LLM** — fast, cheap, deterministic, auditable.
|
|
40
|
+
(A data read still pays one extraction call; only writes and navigate-only reads are 0-LLM end to end.)
|
|
41
|
+
|
|
42
|
+
## What it's for, and what it never became
|
|
43
|
+
|
|
44
|
+
**Good fit** — *repeated* browser automation: scheduled data pulls from authenticated dashboards,
|
|
45
|
+
internal tooling, portal extractions where the flow is stable and run often. *"Every morning, log
|
|
46
|
+
into the vendor portal and pull yesterday's order count."* Learning a flow costs a handful of model
|
|
47
|
+
calls and cents — median **4 calls** and **$0.07** over the 87 measured learns in the committed
|
|
48
|
+
benchmark series (range 2–13 calls, $0.03–$0.18). Each later run re-drives the recipe with no
|
|
49
|
+
planning model in the loop. For an end-to-end wall-clock figure, run the shipped example: on
|
|
50
|
+
2026-09-09 `examples/hn_digest.py` learned in 30 s and replayed in 5 s against a live site — one
|
|
51
|
+
flow, one run, so treat it as an illustration rather than a benchmark.
|
|
52
|
+
|
|
53
|
+
**Not a fit** — a no-code "do anything" agent; one-off complex analysis (use a per-step LLM agent
|
|
54
|
+
for those); anything high-stakes run unsupervised without a human verifying the learned flow. These
|
|
55
|
+
said *not yet* until 0.181.0; development stopped, so they are simply not a fit.
|
|
56
|
+
|
|
57
|
+
### What this is now — the scope the measurements actually support
|
|
58
|
+
|
|
59
|
+
Read this before adopting it. The claims above hold **inside a scope that has been measured on two
|
|
60
|
+
real applications**, and the honest boundary is narrower than the feature list below:
|
|
61
|
+
|
|
62
|
+
- **Supervised authoring, unattended replay.** A human reviews the learned recipe and approves it;
|
|
63
|
+
the approval is bound to a digest of the steps reviewed, so a re-authored recipe refuses rather
|
|
64
|
+
than running. Reads then run unattended. A **declared** write (`spec.mutate` set) sits behind a
|
|
65
|
+
human checkpoint — `flow dry-run` → `flow inspect` → `flow approve`. A write the classifier merely
|
|
66
|
+
*infers* is gated, idempotency-keyed and refused a re-author, but is **not** approval-gated.
|
|
67
|
+
- **Measured on:** a server-rendered app (Gitea 1.22) and one client-rendered SPA (Odoo 17 CRM),
|
|
68
|
+
seven scenarios each, three passes each. `availability_rate` **0.762** (Gitea, cut 2026-08-26) and
|
|
69
|
+
**0.714** (Odoo, cut 2026-09-05 from a series run at 0.169.0), each over n=21 scenario-observations.
|
|
70
|
+
**Neither has been re-cut since**, and neither file records its own cut date or version. A later
|
|
71
|
+
Gitea series measured 0.857 and was **never cut as a baseline**, so that figure has no committed
|
|
72
|
+
artifact behind it — read all three as a level, not a current rate. No silently-wrong outcome is recorded in any committed
|
|
73
|
+
customer-benchmark series (`inviolable: []` throughout); note that `drift_bench`, a different
|
|
74
|
+
instrument, publishes its own non-zero wrong-bind allowlist. The rows that do not pass are named
|
|
75
|
+
and diagnosed in [baselines/README.md](https://github.com/raimondasl/ultracua/blob/main/baselines/README.md).
|
|
76
|
+
- **Expect first-run engineering on a new application.** Reaching that number on the SPA took ten
|
|
77
|
+
`src/` fixes across ~40 releases (0.133.0 → 0.172.0): render readiness, reads served over POST,
|
|
78
|
+
off-screen controls, an element cap, a scroll that landed on the wrong container. Most are general
|
|
79
|
+
and now help both apps — but a third kind of app should be assumed to need its own.
|
|
80
|
+
- **Fails loud, by design:** anti-bot / CAPTCHA interstitials (the run escalates rather than burning
|
|
81
|
+
retries), an ambiguous locator match, a write whose form scope drifted (never LLM-healed), a login
|
|
82
|
+
the built-in form-filler cannot complete, and a WebSocket frame seen during `flow record` or
|
|
83
|
+
`flow dry-run`.
|
|
84
|
+
- **Not supported — and these are gaps, not refusals.** 2FA / SSO logins need a user-supplied
|
|
85
|
+
callable and are not first-class. **iframe and shadow-DOM** content is not captured at all (top
|
|
86
|
+
frame only), and a sub-frame write is deliberately excluded from write reconciliation, so an iframe
|
|
87
|
+
write triggered by a recorded action can cache **ungated**. A write issued from a **shared worker**
|
|
88
|
+
is invisible to every request watcher Playwright offers (`R4.154`, open) — it can double-submit at
|
|
89
|
+
learn and cache as a read. Reads served over **GraphQL POST** are still classified as writes
|
|
90
|
+
(`R4.27`, open); through `replay()` that is a loud refusal, but through a `mode="auto"` door it
|
|
91
|
+
quietly falls back to a re-author and you lose the 0-LLM replay.
|
|
92
|
+
|
|
93
|
+
It's a **usable prototype of a real pattern, not a turnkey product.** The honest status, the measured
|
|
94
|
+
benchmark numbers, and the known gaps live in **[STATUS.md](https://github.com/raimondasl/ultracua/blob/main/STATUS.md)**.
|
|
95
|
+
|
|
96
|
+
## Install
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
pip install ultracua
|
|
100
|
+
python -m playwright install chromium # one-time browser download; nothing works without it
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Optional extras, for the surfaces the core does not pull:
|
|
104
|
+
|
|
105
|
+
```bash
|
|
106
|
+
pip install "ultracua[providers]" # the OpenAI and Gemini adapters
|
|
107
|
+
pip install "ultracua[mcp]" # `ultracua flow serve-mcp`
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Python **3.12** — the only version this has ever been tested on. The dependency floors are capped at
|
|
111
|
+
the next major on purpose: nothing will update this package again, and an uncapped floor is a
|
|
112
|
+
guarantee of eventual breakage rather than of future compatibility.
|
|
113
|
+
|
|
114
|
+
### Working on the source instead
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
uv sync --all-groups # create the venv + install deps (uv manages Python too)
|
|
118
|
+
# NOT bare `uv sync` — it strips the bench/providers/mcp
|
|
119
|
+
# groups and the test suite then fails to import
|
|
120
|
+
uv run playwright install chromium # one-time browser download
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
You'll need `ANTHROPIC_API_KEY` (e.g. in a gitignored `.env`) for the one-time *learn* run and the
|
|
124
|
+
per-run extraction call.
|
|
125
|
+
|
|
126
|
+
## Quickstart
|
|
127
|
+
|
|
128
|
+
Define a recurring task once, learn it, then replay it with 0-LLM navigation — it returns structured data and
|
|
129
|
+
**fails loud** on drift instead of returning a wrong value:
|
|
130
|
+
|
|
131
|
+
```python
|
|
132
|
+
import asyncio
|
|
133
|
+
from ultracua import FlowSpec, learn_flow, approve_flow, replay_flow
|
|
134
|
+
|
|
135
|
+
spec = FlowSpec(
|
|
136
|
+
name="daily-orders",
|
|
137
|
+
start_url="https://portal.example.com/admin",
|
|
138
|
+
goal="open the orders report",
|
|
139
|
+
extract="the number of orders placed yesterday", # → structured data
|
|
140
|
+
)
|
|
141
|
+
res = asyncio.run(learn_flow(spec)) # author once; inspect res.steps / res.data
|
|
142
|
+
approve_flow(spec) # a human verifies before trusting it unattended
|
|
143
|
+
data = asyncio.run(replay_flow(spec)) # every run after: 0-LLM navigation, returns the data
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
Prefer a real, runnable walkthrough? **[EXAMPLES.md](https://github.com/raimondasl/ultracua/blob/main/EXAMPLES.md)** does this end-to-end against
|
|
147
|
+
Hacker News (read-only) and is built to record: `uv run python examples/hn_digest.py --headed`.
|
|
148
|
+
|
|
149
|
+
## Highlights
|
|
150
|
+
|
|
151
|
+
- **0-LLM navigation** — through `replay_flow()` / `flow replay` (`mode="replay"`) a learned flow re-drives every step with no model call; this is pinned key-lessly by `tests/test_inviolable_properties.py`, which sweeps the drift corpus with a floor of 90 rows (99 on the last run) and asserts per row that replay resolves no provider at all. A write with no readback is then 0-LLM end to end (0 calls on all 24 write replays in the committed series — see the headline for what those 24 contain); a *data* read, **and any write that sets `extract=`**, adds one extraction call — measured **1 call on 43 of 51 read replays, 2 on the other 8**, where an auth-refresh retry re-extracts. `pin_read` would remove even that, but it only fires for a scalar answer that maps to exactly one element carrying an `id`/`data-testid`: measured pinnable on **0 of the corpus's 10 read scenarios**, so budget one call per read.
|
|
152
|
+
- **Resilient, self-healing locators** — survive cosmetic DOM drift; one-step LLM re-grounding on real drift, or a **suffix-replan** that re-authors just the broken tail (keeping the working prefix) when the path changes. **Measured**, key-lessly and in CI on a 187-row corpus: 0-LLM survival degrades with mutation intensity and reaches zero — 20/27 at k=1, 0/6 at k=7, non-monotonic in between (k50 = 6) — and the heal machinery recovers **36 of 39** heal-eligible rows *given* correct element identity ([drift-bench v2](https://github.com/raimondasl/ultracua/blob/main/baselines/README.md) — read the limits; that heal figure is a mechanism ceiling measured against a perfect-vision oracle, not a heal rate; the committed `drift_v2.json` was two releases stale when this line first said so and was re-recorded at 0.181.0, R4.159). Two wrong-bind classes are **published rather than hidden**, each pinned by its own corpus row.
|
|
153
|
+
- **Trust controls** — an approval gate **bound to the steps you reviewed** (re-author them and replay refuses, before the browser opens), data-shape + value-contract drift detection, **fail-loud** `FlowReplayError`.
|
|
154
|
+
- **Auth refresh** — re-login on session expiry; credentials are env-sourced and **never persisted**.
|
|
155
|
+
- **Write flows** — submit / post / purchase with **action-completion verification** + idempotency.
|
|
156
|
+
- **Dry run** — `flow dry-run` replays a write flow with **every write held**: see the exact body, URL and idempotency key it *would* send before you approve it. Anything that cannot be *proven* held aborts rather than proceeding — **within the realms a watcher can see**. A write issued from a **shared worker** is invisible to every request watcher Playwright offers (`R4.154`, open, measured on both benchmark substrates), so for that realm neither the hold nor the abort can fire.
|
|
157
|
+
- **Record by demonstration** — `ultracua flow record` captures a headed walkthrough into a cached **0-LLM** flow: reads are verify-by-replay; declared writes are **gated + approval-gated + idempotency-keyed**.
|
|
158
|
+
- **Fleet supervisor** — `flow run-all` replays every saved flow, reports pass/fail, alerts, exits non-zero for cron — including when a flow was refused a run by something no human chose, or when nothing ran at all; `flow status` for history.
|
|
159
|
+
- **Multi-provider** — Anthropic / OpenAI / Gemini, fast/strong tiering, prompt caching.
|
|
160
|
+
- **Drive from any language** — JSON-RPC daemon + a Node/JS client.
|
|
161
|
+
|
|
162
|
+
## Documentation
|
|
163
|
+
|
|
164
|
+
| Doc | For |
|
|
165
|
+
|---|---|
|
|
166
|
+
| **[EXAMPLES.md](https://github.com/raimondasl/ultracua/blob/main/EXAMPLES.md)** | a worked, runnable real-site example — **start here** |
|
|
167
|
+
| **[GUIDE.md](https://github.com/raimondasl/ultracua/blob/main/GUIDE.md)** | developer guide: the Flow API + CLI in depth (auth, write flows, record by demonstration, health, providers) |
|
|
168
|
+
| **[HEALING.md](https://github.com/raimondasl/ultracua/blob/main/HEALING.md)** | how it self-heals (and deliberately doesn't) when a page's elements change: resilient locators, LLM heal/re-plan, and the fail-loud boundaries |
|
|
169
|
+
| **[docs/comparison-stagehand.md](https://github.com/raimondasl/ultracua/blob/main/docs/comparison-stagehand.md)** | ultracua vs. Stagehand — a design-philosophy comparison (drift/self-heal, write safety, data correctness), dated + sourced |
|
|
170
|
+
| **[docs/open-defects.md](https://github.com/raimondasl/ultracua/blob/main/docs/open-defects.md)** | the standing defect register — **four** adversarial rounds. Rounds 1–2 are fixed; of the 13 R3-numbered entries (round 3 found 11, R3.12–R3.13 were filed later), **2 remain open** (R3.2, R3.7); round 4 began as a pre-merge audit that *parked* a change rather than ship it and became a standing per-slice log — **160 findings, 73 open / 83 fixed / 4 parked**. Records the residuals it decided not to fix, including the decisions closed as NO CHANGE. Its R4 index is rendered from `docs/register/state.json` and its R3 count is parsed from the headings — both machine-checked; **this table row is prose and is not**, which is why it said "nine remain open" for five weeks after that stopped being true. Check the register's own index, not this row. **Read before starting new work.** |
|
|
171
|
+
| **[docs/correctness-plan.md](https://github.com/raimondasl/ultracua/blob/main/docs/correctness-plan.md)** | **closed at 0.181.0** — the plan behind Phases 0–7, grounded on the v0.75.0 survey (round-3 findings, test-machinery holes, unpinned residuals) — worst user harm first, test-first, one slice per PR. It does **not** sequence the round-4 register: most of the 73 open findings post-date it, and `docs/reshape-plan.md` §13 plus `docs/plan/state.json` are the operative order. |
|
|
172
|
+
| **[docs/correctness-survey.md](https://github.com/raimondasl/ultracua/blob/main/docs/correctness-survey.md)** | the measured inventory the plan is built on: **59 items** across the register, CI/eval machinery, the user-facing surface and the accepted residuals in `src/`, produced 2026-08-04 at v0.75.0 and not re-derived since. |
|
|
173
|
+
| **[ARCHITECTURE.md](https://github.com/raimondasl/ultracua/blob/main/ARCHITECTURE.md)** | how it works inside + how to contribute (engine, safety, tiers, benchmarks, layout) |
|
|
174
|
+
| **[STATUS.md](https://github.com/raimondasl/ultracua/blob/main/STATUS.md)** | honest status, measured benchmarks, known fragilities |
|
|
175
|
+
| **[ROADMAP.md](https://github.com/raimondasl/ultracua/blob/main/ROADMAP.md)** | what was *going* to be next — **closed at 0.181.0**, kept as the record of what was intended. Its opening "2–7×" claim is unsupported and is corrected in its own banner |
|
|
176
|
+
| **[evals/README.md](https://github.com/raimondasl/ultracua/blob/main/evals/README.md)** | the manual capability-eval suite: shipped behavior + the H1–H16 horizons, with $ cost estimates and partial runs |
|
|
177
|
+
| **[PLAN.md](https://github.com/raimondasl/ultracua/blob/main/PLAN.md)** | the original design + research basis |
|
|
178
|
+
| **[docs/recorder-spike.md](https://github.com/raimondasl/ultracua/blob/main/docs/recorder-spike.md)** | the record-by-demonstration design (capture → gated cache → 0-LLM replay) |
|
|
179
|
+
|
|
180
|
+
## License
|
|
181
|
+
|
|
182
|
+
[MIT](https://github.com/raimondasl/ultracua/blob/main/LICENSE) © Raimondas Lencevicius
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "ultracua"
|
|
3
|
+
version = "0.182.0"
|
|
4
|
+
description = "A Computer Use Agent that learns a browser flow once with an LLM, then replays it deterministically with no planning model in the loop."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
license-files = ["LICENSE"]
|
|
8
|
+
requires-python = ">=3.12"
|
|
9
|
+
authors = [{ name = "Raimondas Lencevicius" }]
|
|
10
|
+
keywords = [
|
|
11
|
+
"browser-automation", "computer-use-agent", "playwright", "llm-agent",
|
|
12
|
+
"deterministic-replay", "web-scraping", "rpa",
|
|
13
|
+
]
|
|
14
|
+
# `Development Status :: 7 - Inactive` IS THE LOAD-BEARING ONE. PyPI renders it as a badge, and it is
|
|
15
|
+
# the only MACHINE-READABLE way to say what the README's banner says in prose: active development
|
|
16
|
+
# stopped on 2026-09-09. A package that omits it reads as maintained BY DEFAULT, which is the
|
|
17
|
+
# quiet-wrong direction this project spent four audit rounds refusing everywhere else.
|
|
18
|
+
classifiers = [
|
|
19
|
+
"Development Status :: 7 - Inactive",
|
|
20
|
+
"Intended Audience :: Developers",
|
|
21
|
+
"Operating System :: OS Independent",
|
|
22
|
+
"Programming Language :: Python :: 3",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
# 3.13 is DELIBERATELY ABSENT. CI runs 3.12 on both OSes and `.python-version` is 3.12, so
|
|
25
|
+
# this package has never been executed on 3.13. Claiming it would be an unmeasured support
|
|
26
|
+
# claim in permanent metadata, which is the one thing this release exists to avoid. For the
|
|
27
|
+
# same reason `requires-python` stays >=3.12 rather than widening to 3.11: dropping the floor
|
|
28
|
+
# would be an unmeasured widening, and nothing will ever test it.
|
|
29
|
+
"Topic :: Internet :: WWW/HTTP :: Browsers",
|
|
30
|
+
"Topic :: Software Development :: Testing",
|
|
31
|
+
]
|
|
32
|
+
# UPPER BOUNDS ARE NOT OPTIONAL ON AN ARCHIVED PACKAGE, and this one was measured rather than
|
|
33
|
+
# feared. With `anthropic>=0.109.2` unbounded, a clean `pip install ultracua` on 2026-09-10 resolves
|
|
34
|
+
# to **anthropic 1.4.0**, whose `messages.create` no longer accepts `temperature` -- and
|
|
35
|
+
# `providers/llm_agent.py:131` passes it on every authoring call. Reproduced against the installed
|
|
36
|
+
# wheel: `TypeError: AsyncMessages.create() got an unexpected keyword argument 'temperature'`, raised
|
|
37
|
+
# BEFORE any network call. So every learn run of a freshly-installed 0.182.0 would have died on its
|
|
38
|
+
# first model call, which is the product's primary path.
|
|
39
|
+
#
|
|
40
|
+
# `uv.lock` pins anthropic 0.109.2, and that is the ONLY version this suite has ever exercised. The
|
|
41
|
+
# caps are therefore the next MAJOR in each case -- the boundary where a removal like that belongs --
|
|
42
|
+
# and they are a statement about what was tested, not a prediction about what will break. A package
|
|
43
|
+
# that will never be updated again cannot rely on "upstream stays compatible"; nothing is watching.
|
|
44
|
+
dependencies = [
|
|
45
|
+
"anthropic>=0.109.2,<1.0",
|
|
46
|
+
"playwright>=1.60.0,<2.0",
|
|
47
|
+
"pydantic>=2.13.4,<3.0",
|
|
48
|
+
"python-dotenv>=1.2.2,<2.0",
|
|
49
|
+
"xxhash>=3.7.0,<5.0",
|
|
50
|
+
]
|
|
51
|
+
|
|
52
|
+
# WHY THESE EXIST AT ALL, since the package is archived: a reader who lands on the PyPI page must be
|
|
53
|
+
# able to REACH `docs/open-defects.md`. The close-out decision (2026-09-09) was to publish the METHOD
|
|
54
|
+
# rather than the product, and the method lives in the register and in CLAUDE.md -- neither of which
|
|
55
|
+
# ships inside the sdist. Without a Project-URL the page is a dead end and the one artifact worth
|
|
56
|
+
# reading is unreachable from the one place a stranger will find this.
|
|
57
|
+
# EXTRAS EXIST BECAUSE `[dependency-groups]` BELOW DOES NOT REACH A pip USER. Those groups are
|
|
58
|
+
# PEP-735, which uv understands and `pip install` does not -- they produce no `Provides-Extra` in the
|
|
59
|
+
# wheel metadata at all. So README's "Multi-provider -- Anthropic / OpenAI / Gemini" and the MCP
|
|
60
|
+
# server were advertised on the PyPI page with no way to install their dependencies: a pip user
|
|
61
|
+
# reaching `providers/openai.py` got a bare `ModuleNotFoundError` naming no remedy. The groups stay
|
|
62
|
+
# for the development workflow; these mirror the two a CONSUMER needs.
|
|
63
|
+
[project.optional-dependencies]
|
|
64
|
+
providers = ["google-genai>=2.8.0,<3.0", "openai>=2.42.0,<3.0"]
|
|
65
|
+
mcp = ["mcp>=1.28.0,<2.0"]
|
|
66
|
+
|
|
67
|
+
[project.urls]
|
|
68
|
+
Repository = "https://github.com/raimondasl/ultracua"
|
|
69
|
+
"Defect register" = "https://github.com/raimondasl/ultracua/blob/main/docs/open-defects.md"
|
|
70
|
+
"Honest status" = "https://github.com/raimondasl/ultracua/blob/main/STATUS.md"
|
|
71
|
+
"Benchmark numbers" = "https://github.com/raimondasl/ultracua/blob/main/baselines/README.md"
|
|
72
|
+
|
|
73
|
+
[project.scripts]
|
|
74
|
+
ultracua = "ultracua:main"
|
|
75
|
+
ultracua-daemon = "ultracua.daemon:main"
|
|
76
|
+
|
|
77
|
+
[build-system]
|
|
78
|
+
requires = ["uv_build>=0.11.21,<0.12.0"]
|
|
79
|
+
build-backend = "uv_build"
|
|
80
|
+
|
|
81
|
+
[tool.pytest.ini_options]
|
|
82
|
+
asyncio_mode = "auto"
|
|
83
|
+
testpaths = ["tests"]
|
|
84
|
+
|
|
85
|
+
[dependency-groups]
|
|
86
|
+
bench = [
|
|
87
|
+
"miniwob>=1.0",
|
|
88
|
+
]
|
|
89
|
+
dev = [
|
|
90
|
+
"pytest>=9.1.0",
|
|
91
|
+
"pytest-asyncio>=1.4.0",
|
|
92
|
+
# CI shards the suite across runners (see .github/workflows/ci.yml). It PARTITIONS THE REAL
|
|
93
|
+
# COLLECTION, which is the property that matters: a test file added tomorrow lands in some shard by
|
|
94
|
+
# construction. A hand-maintained file list would let a new file run in NEITHER while both shards
|
|
95
|
+
# report green — the "quiet is an allowlist" failure wearing a CI hat.
|
|
96
|
+
"pytest-split>=0.10.0",
|
|
97
|
+
]
|
|
98
|
+
providers = [
|
|
99
|
+
"google-genai>=2.8.0",
|
|
100
|
+
"openai>=2.42.0",
|
|
101
|
+
]
|
|
102
|
+
# H2 flows-as-tools: the MCP server (`ultracua flow serve-mcp`). Optional — the core package and the
|
|
103
|
+
# mcpserver module import without it; only actually serving needs the SDK (`uv sync --group mcp`).
|
|
104
|
+
mcp = [
|
|
105
|
+
"mcp>=1.28.0",
|
|
106
|
+
]
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""ultracua — a Computer Use Agent that learns a browser flow once with an LLM, then replays it
|
|
2
|
+
deterministically with no model in the loop.
|
|
3
|
+
|
|
4
|
+
This said *drives a browser at 5-10x human speed* until 0.182.0. Nothing in this tree measures that:
|
|
5
|
+
what is committed is what replay REMOVES (0 model calls on all 24 write replays in the benchmark
|
|
6
|
+
series, one on 43 of 51 read replays) plus `baselines/demo.json`'s FIXTURE speedup of 62-114x, which
|
|
7
|
+
is removed model latency against a LOCAL page and does not transfer to a real site. See `README.md`.
|
|
8
|
+
|
|
9
|
+
ACTIVE DEVELOPMENT STOPPED ON 2026-09-09. The defect register carries **75 open findings** and a
|
|
10
|
+
close-out banner explaining what that means; read it before adopting this. It is not in this package
|
|
11
|
+
-- the sdist ships only `src/`, the LICENSE and this README -- so it is linked from the PyPI page and
|
|
12
|
+
lives at https://github.com/raimondasl/ultracua/blob/main/docs/open-defects.md
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from .agent import run_goal
|
|
18
|
+
from .browser import BrowserSession
|
|
19
|
+
from .cache import CachedFlow, CachedStep, FlowCache, flow_key
|
|
20
|
+
from .config import settings
|
|
21
|
+
from .flow import FlowReport, run_cached
|
|
22
|
+
from .locators import LocatorSpec
|
|
23
|
+
from .parallel import run_many
|
|
24
|
+
from .safety import PacingGovernor, is_mutating
|
|
25
|
+
from .types import Action, Element, Observation, StepResult
|
|
26
|
+
from .verifiers import keyword_completion
|
|
27
|
+
from .extract import Extraction, extract
|
|
28
|
+
from .flows import (
|
|
29
|
+
FleetRun, FlowHealth, FlowQuarantineError, FlowReplayError, FlowSpec, LoginSpec, MutateSpec, SlotSpec,
|
|
30
|
+
WriteReadbackError, WriteUnverifiedError, refresh_auth,
|
|
31
|
+
)
|
|
32
|
+
from .flows import AuditFinding, AuditRun
|
|
33
|
+
from .flows import approve as approve_flow
|
|
34
|
+
from .flows import audit_flows
|
|
35
|
+
from .flows import health as flow_health
|
|
36
|
+
from .flows import learn as learn_flow
|
|
37
|
+
from .flows import release as release_flow
|
|
38
|
+
from .flows import replay as replay_flow
|
|
39
|
+
from .flows import run_all as run_all_flows
|
|
40
|
+
from .flows import unapprove as unapprove_flow
|
|
41
|
+
from .vision import AnthropicGrounding, MockGrounding
|
|
42
|
+
|
|
43
|
+
# Single-source the version from the installed package metadata (pyproject.toml is the source of
|
|
44
|
+
# truth). Falls back for an uninstalled source checkout.
|
|
45
|
+
try:
|
|
46
|
+
from importlib.metadata import PackageNotFoundError, version as _pkg_version
|
|
47
|
+
|
|
48
|
+
__version__ = _pkg_version("ultracua")
|
|
49
|
+
except (PackageNotFoundError, ImportError): # pragma: no cover - source-tree fallback
|
|
50
|
+
__version__ = "0.0.0+dev"
|
|
51
|
+
|
|
52
|
+
__all__ = [
|
|
53
|
+
"BrowserSession",
|
|
54
|
+
"Action",
|
|
55
|
+
"Element",
|
|
56
|
+
"Observation",
|
|
57
|
+
"StepResult",
|
|
58
|
+
"LocatorSpec",
|
|
59
|
+
"CachedFlow",
|
|
60
|
+
"CachedStep",
|
|
61
|
+
"FlowCache",
|
|
62
|
+
"flow_key",
|
|
63
|
+
"FlowReport",
|
|
64
|
+
"PacingGovernor",
|
|
65
|
+
"is_mutating",
|
|
66
|
+
"run_goal",
|
|
67
|
+
"run_cached",
|
|
68
|
+
"run_many",
|
|
69
|
+
"keyword_completion",
|
|
70
|
+
"AnthropicGrounding",
|
|
71
|
+
"MockGrounding",
|
|
72
|
+
"FlowSpec",
|
|
73
|
+
"LoginSpec",
|
|
74
|
+
"MutateSpec",
|
|
75
|
+
"SlotSpec",
|
|
76
|
+
"FlowHealth",
|
|
77
|
+
"FleetRun",
|
|
78
|
+
"FlowReplayError",
|
|
79
|
+
"FlowQuarantineError",
|
|
80
|
+
"WriteReadbackError",
|
|
81
|
+
"WriteUnverifiedError",
|
|
82
|
+
"learn_flow",
|
|
83
|
+
"replay_flow",
|
|
84
|
+
"approve_flow",
|
|
85
|
+
"unapprove_flow",
|
|
86
|
+
"release_flow",
|
|
87
|
+
"audit_flows",
|
|
88
|
+
"AuditRun",
|
|
89
|
+
"AuditFinding",
|
|
90
|
+
"run_all_flows",
|
|
91
|
+
"refresh_auth",
|
|
92
|
+
"flow_health",
|
|
93
|
+
"extract",
|
|
94
|
+
"Extraction",
|
|
95
|
+
"settings",
|
|
96
|
+
"main",
|
|
97
|
+
]
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def main() -> None:
|
|
101
|
+
from .cli import main as _main
|
|
102
|
+
|
|
103
|
+
_main()
|