mio-cua 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mio_cua-0.1.0/PKG-INFO +13 -0
- mio_cua-0.1.0/README.md +110 -0
- mio_cua-0.1.0/mio_cua/__init__.py +14 -0
- mio_cua-0.1.0/mio_cua/agent/__init__.py +0 -0
- mio_cua-0.1.0/mio_cua/agent/diff.py +81 -0
- mio_cua-0.1.0/mio_cua/agent/dispatcher.py +39 -0
- mio_cua-0.1.0/mio_cua/agent/expected.py +48 -0
- mio_cua-0.1.0/mio_cua/agent/loop.py +358 -0
- mio_cua-0.1.0/mio_cua/agent/planner.py +176 -0
- mio_cua-0.1.0/mio_cua/agent/recover.py +52 -0
- mio_cua-0.1.0/mio_cua/agent/safety.py +68 -0
- mio_cua-0.1.0/mio_cua/agent/verify.py +23 -0
- mio_cua-0.1.0/mio_cua/agent_factory.py +72 -0
- mio_cua-0.1.0/mio_cua/automation/__init__.py +0 -0
- mio_cua-0.1.0/mio_cua/automation/backends.py +161 -0
- mio_cua-0.1.0/mio_cua/automation/input_controller.py +48 -0
- mio_cua-0.1.0/mio_cua/automation/uia.py +48 -0
- mio_cua-0.1.0/mio_cua/automation/windows.py +315 -0
- mio_cua-0.1.0/mio_cua/cli.py +179 -0
- mio_cua-0.1.0/mio_cua/config.py +37 -0
- mio_cua-0.1.0/mio_cua/events.py +38 -0
- mio_cua-0.1.0/mio_cua/llm/__init__.py +0 -0
- mio_cua-0.1.0/mio_cua/llm/client.py +23 -0
- mio_cua-0.1.0/mio_cua/mcp_server.py +174 -0
- mio_cua-0.1.0/mio_cua/memory/__init__.py +5 -0
- mio_cua-0.1.0/mio_cua/memory/artifact.py +96 -0
- mio_cua-0.1.0/mio_cua/memory/history.py +12 -0
- mio_cua-0.1.0/mio_cua/memory/state.py +23 -0
- mio_cua-0.1.0/mio_cua/models/__init__.py +11 -0
- mio_cua-0.1.0/mio_cua/models/action.py +23 -0
- mio_cua-0.1.0/mio_cua/models/action_result.py +18 -0
- mio_cua-0.1.0/mio_cua/models/element.py +16 -0
- mio_cua-0.1.0/mio_cua/models/observation.py +27 -0
- mio_cua-0.1.0/mio_cua/models/task.py +21 -0
- mio_cua-0.1.0/mio_cua/perception/__init__.py +4 -0
- mio_cua-0.1.0/mio_cua/perception/merger.py +77 -0
- mio_cua-0.1.0/mio_cua/perception/perception.py +146 -0
- mio_cua-0.1.0/mio_cua/prompts/__init__.py +11 -0
- mio_cua-0.1.0/mio_cua/prompts/system.txt +35 -0
- mio_cua-0.1.0/mio_cua/providers/__init__.py +0 -0
- mio_cua-0.1.0/mio_cua/providers/base.py +15 -0
- mio_cua-0.1.0/mio_cua/providers/openai_compat.py +38 -0
- mio_cua-0.1.0/mio_cua/scene/__init__.py +60 -0
- mio_cua-0.1.0/mio_cua/scene/affordances.py +221 -0
- mio_cua-0.1.0/mio_cua/scene/builder.py +140 -0
- mio_cua-0.1.0/mio_cua/scene/diff.py +86 -0
- mio_cua-0.1.0/mio_cua/scene/graph.py +102 -0
- mio_cua-0.1.0/mio_cua/scene/memory.py +67 -0
- mio_cua-0.1.0/mio_cua/scene/omniparser.py +127 -0
- mio_cua-0.1.0/mio_cua/scene/regions.py +89 -0
- mio_cua-0.1.0/mio_cua/scene/relations.py +109 -0
- mio_cua-0.1.0/mio_cua/simulation.py +272 -0
- mio_cua-0.1.0/mio_cua/tools/__init__.py +0 -0
- mio_cua-0.1.0/mio_cua/tools/builtin.py +58 -0
- mio_cua-0.1.0/mio_cua/tools/click.py +14 -0
- mio_cua-0.1.0/mio_cua/tools/context.py +14 -0
- mio_cua-0.1.0/mio_cua/tools/fail.py +7 -0
- mio_cua-0.1.0/mio_cua/tools/focus_window.py +10 -0
- mio_cua-0.1.0/mio_cua/tools/fs.py +88 -0
- mio_cua-0.1.0/mio_cua/tools/key.py +7 -0
- mio_cua-0.1.0/mio_cua/tools/launch.py +111 -0
- mio_cua-0.1.0/mio_cua/tools/move_mouse.py +14 -0
- mio_cua-0.1.0/mio_cua/tools/registry.py +21 -0
- mio_cua-0.1.0/mio_cua/tools/screenshot.py +18 -0
- mio_cua-0.1.0/mio_cua/tools/scroll.py +7 -0
- mio_cua-0.1.0/mio_cua/tools/success.py +7 -0
- mio_cua-0.1.0/mio_cua/tools/type.py +27 -0
- mio_cua-0.1.0/mio_cua/tools/wait.py +8 -0
- mio_cua-0.1.0/mio_cua/vision/__init__.py +0 -0
- mio_cua-0.1.0/mio_cua/vision/ocr.py +47 -0
- mio_cua-0.1.0/mio_cua/vision/overlay.py +13 -0
- mio_cua-0.1.0/mio_cua/vision/screen.py +39 -0
- mio_cua-0.1.0/mio_cua.egg-info/PKG-INFO +13 -0
- mio_cua-0.1.0/mio_cua.egg-info/SOURCES.txt +78 -0
- mio_cua-0.1.0/mio_cua.egg-info/dependency_links.txt +1 -0
- mio_cua-0.1.0/mio_cua.egg-info/entry_points.txt +3 -0
- mio_cua-0.1.0/mio_cua.egg-info/requires.txt +8 -0
- mio_cua-0.1.0/mio_cua.egg-info/top_level.txt +1 -0
- mio_cua-0.1.0/pyproject.toml +33 -0
- mio_cua-0.1.0/setup.cfg +4 -0
mio_cua-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mio-cua
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Mio Computer-Use Agent: Windows desktop automation AI agent (SDK + CLI)
|
|
5
|
+
Requires-Python: >=3.10
|
|
6
|
+
Requires-Dist: requests>=2.31
|
|
7
|
+
Requires-Dist: mss>=9.0
|
|
8
|
+
Requires-Dist: Pillow>=10.0
|
|
9
|
+
Requires-Dist: PyYAML>=6.0
|
|
10
|
+
Requires-Dist: pynput>=1.7
|
|
11
|
+
Requires-Dist: pywin32>=306
|
|
12
|
+
Requires-Dist: pywinauto>=0.6.8
|
|
13
|
+
Requires-Dist: rapidocr_onnxruntime>=1.3.24
|
mio_cua-0.1.0/README.md
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
# mio-cua
|
|
2
|
+
|
|
3
|
+
Mio Computer-Use Agent:Windows 桌面自动化 AI Agent(Python SDK + CLI)。
|
|
4
|
+
|
|
5
|
+
<!-- mcp-name: io.github.mldlbs/mio-cua -->
|
|
6
|
+
|
|
7
|
+
## 安装
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install -e . # 从源码安装(依赖见 pyproject.toml)
|
|
11
|
+
# Windows 下建议以管理员身份运行
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
### 配置
|
|
15
|
+
|
|
16
|
+
设置 LLM API Key:
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
set OPENAI_API_KEY=sk-xxx # Windows CMD
|
|
20
|
+
$env:OPENAI_API_KEY = "sk-xxx" # PowerShell
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
示例脚本还支持 `DESKTOP_AGENT_PROVIDER` / `DESKTOP_AGENT_MODEL` 环境变量覆盖模型。
|
|
24
|
+
|
|
25
|
+
### 当前状态
|
|
26
|
+
|
|
27
|
+
v0.5:Scene Graph 感知层 + 网页纯视觉感知(Regions 版面 + OmniParser 控件识别);五个端到端场景(记事本/计算器/资源管理器/跨应用/网页)已在真实 Windows 11 桌面完成 5/5 全绿终验(2026-08-08)。
|
|
28
|
+
|
|
29
|
+
- **Scene Graph 感知层**:感知层把 OCR+UIA 融合为一张场景图(`mio_cua/scene/`)——每个 UI 对象是一个 Node(类型/文本/状态/bbox),配空间关系(leftOf/above/labelFor)和 **Action candidates**(感知层已验证的可点击动作,如 `click node 7 ('7') {'value': '7'} expects {'display': True}`)。LLM 从候选动作里选择,不再自行推断坐标/UI。
|
|
30
|
+
- **网页纯视觉感知(像人一样上网)**:浏览器窗口额外走两条视觉链路,不依赖 DOM——
|
|
31
|
+
- **Regions 版面分析**(`rapid_layout`,可选依赖):识别页面结构(导航/标题/正文/表格/图片区域),可选依赖未装则优雅降级。
|
|
32
|
+
- **OmniParser 控件识别**(`scene/omniparser.py`,可选依赖):YOLOv9 交互元素检测 + Florence-2 语义描述,把网页截图解析成按钮/链接/输入框(`button`/`text` 节点 + 可点击候选)。需 torch CUDA;模型默认在项目内 `models/omniparser/`,也可用 `OMNIPARSER_DIR`/`OMNIPARSER_WEIGHTS` 环境变量覆盖。
|
|
33
|
+
- **显示区验证**:计算器显示区被识别为 `display` 节点,点击数字键后通过 Scene Diff 检测显示值变化(`0`→`7`),用于动作成功校验。
|
|
34
|
+
- **跨应用(crossapp)场景**:`smoke/crossapp.yaml`——读 `smoke_numbers.txt`(12/34/56)→ 计算器求和 102 → 保存 `smoke_sum_result.txt`。全绿验收(2026-08-08,19-21 步)。关键修复:**一动作一感知**(每步重新感知再决策,避免多动作基于陈旧场景执行);**计算器 `+` 键发送**(`key(keys="+")` 曾被当分隔符 split 失败);**UWP 计算器聚焦**;其余见 SMOKE.md 修复链。
|
|
35
|
+
- **网页纯视觉场景(web)**:`smoke/web.yaml`——Edge 打开本地 HTML,纯视觉(OmniParser,不接 DOM)点击按钮/输入文本并视觉确认。PASS(2026-08-08,9 步 165s)。OmniParser 走本地 HF 缓存离线推理(`HF_HUB_OFFLINE=1`,首帧加载约 20s,后续每帧 ~0.3s)。
|
|
36
|
+
- **完整套件**:calculator/crossapp/explorer/notepad/web 五场景全绿终验(2026-08-08,crossapp 21 步 SUCCESS、其余 SUCCESS/校验通过)。deepseek-v4-flash 即可完成。调优中修复 6 个工程 bug(陈旧场景执行、UWP 聚焦、`+` 键 split、浏览器 PATH、重命名引导、元素 id 稳定)。
|
|
37
|
+
- **GPU OCR**:经 DirectML(onnxruntime-directml)加速,单次 OCR 平均 ~1.6s;`DESKTOP_AGENT_OCR_DEVICE=cpu` 可回退纯 CPU。
|
|
38
|
+
- **窗口区域截图**:只抓活动窗口区域(而非全屏),截图体积缩减约 80%,更聚焦目标窗口。
|
|
39
|
+
- **Artifact 自动清理**:任务结束后自动修剪,目录默认上限 200MB(配置项 `artifact_max_bytes`)。
|
|
40
|
+
- **循环守卫**:动作无变化 / 重复相同动作 ≥6 次直接判定 FAIL,避免空转。
|
|
41
|
+
- **真实验收(deepseek-v4-flash + DirectML)**:记事本(输入并保存 `hello world`)、计算器(`123*456=56088`)、资源管理器(新建并重命名 `smoke_demo_folder`)三个场景全部 PASS。
|
|
42
|
+
|
|
43
|
+
发送给 LLM 的截图带编号框(overlay),编号与元素列表 `id` 一一对应。每步操作与截图会落盘为 artifact,任务状态支持 `resume`。可重试失败会自动触发 Recovery(聚焦窗口后重试,每次 action 最多重试 2 次)。
|
|
44
|
+
|
|
45
|
+
> 提示:首次在真实桌面运行前,请先完成一次小任务的冒烟测试(如"打开记事本输入 hello"),并确认 F9 急停可用。冒烟测试见 `SMOKE.md`。
|
|
46
|
+
|
|
47
|
+
## CLI
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
mio-cua run "打开计算器,计算 3*4" --model gpt-4o
|
|
51
|
+
mio-cua run "删除所有文件" --dry-run # 只打印计划,不执行
|
|
52
|
+
mio-cua resume <task_id> # 从上次状态重新执行任务
|
|
53
|
+
mio-cua replay <task_id> # 从 artifact 回放任务步骤(调试用,--full 显示参数)
|
|
54
|
+
mio-cua providers
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Artifacts / 状态
|
|
58
|
+
|
|
59
|
+
- 每步:`~/.mio_cua/artifacts/<ts>.json`(observation + action + result)
|
|
60
|
+
- 每步截图:`~/.mio_cua/artifacts/<ts>.png`(overlay 标注图)与 `.raw.png`
|
|
61
|
+
- 任务状态:`~/.mio_cua/artifacts/state/<task_id>.json`
|
|
62
|
+
|
|
63
|
+
## SDK
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
from mio_cua import Agent, AgentConfig, Task
|
|
67
|
+
|
|
68
|
+
agent = Agent(AgentConfig(model="gpt-4o", max_steps=50))
|
|
69
|
+
result = agent.run(Task(instruction="打开记事本,输入 hello"))
|
|
70
|
+
print(result.status, result.steps)
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## MCP(接入 Claude / Cursor / ChatGPT)
|
|
74
|
+
|
|
75
|
+
mio-cua 也作为 **MCP server** 暴露,让任意 MCP 客户端直接控制桌面:
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
pip install -e .
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
```json
|
|
82
|
+
{ "mcpServers": { "mio-cua": { "command": "mio-cua-mcp", "args": [] } } }
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
10 个工具:文件操作(list_dir/make_dir/move_file/move_files)、窗口(launch/focus_window/get_active_window)、输入(click/type/key)。详见 [MCP.md](MCP.md)。
|
|
86
|
+
|
|
87
|
+
## 安全
|
|
88
|
+
|
|
89
|
+
- F9 急停
|
|
90
|
+
- 步数上限 / 任务超时
|
|
91
|
+
- 每步截图留痕(artifact)
|
|
92
|
+
|
|
93
|
+
## 验收套件(smoke)
|
|
94
|
+
|
|
95
|
+
五个端到端场景在**独立虚拟桌面**上隔离运行,不打扰主桌面:
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
# 完整套件(calculator/crossapp/explorer/notepad/web)
|
|
99
|
+
python scripts/run_smoke_vdesk.py --only calculator,crossapp,explorer,notepad,web --model deepseek-v4-flash --base-url https://api.deepseek.com/v1
|
|
100
|
+
|
|
101
|
+
# 单场景
|
|
102
|
+
python scripts/run_smoke_vdesk.py --only crossapp --model deepseek-v4-flash --base-url https://api.deepseek.com/v1
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
运行前需:`$env:OPENAI_API_KEY="sk-xxx"`;桌面已解锁;用户约 20 分钟不碰键鼠(vdesk 隔离在用户操作时失效)。日志 `%TEMP%\smoke_vdesk.log`,结果以 `[PASS]/[FAIL]` 汇总。
|
|
106
|
+
|
|
107
|
+
前置文件:`~/Desktop/smoke_numbers.txt`(12/34/56,crossapp 输入)、`~/Desktop/vision_test.html`(web 测试页)。场景间自动清理测试应用进程。
|
|
108
|
+
|
|
109
|
+
- 判定规则:`status=SUCCESS` 且(若场景有产物校验)所有 file/dir/contains 校验通过才算 PASS——防止 agent 假 success。
|
|
110
|
+
- 收尾提示:屏幕稳定且已按 Enter/Save 确认后,提示 agent 调 `success`,避免任务完成却超时。
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
|
2
|
+
|
|
3
|
+
from mio_cua.config import AgentConfig
|
|
4
|
+
from mio_cua.models.task import Task
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def __getattr__(name):
|
|
8
|
+
if name == "Agent":
|
|
9
|
+
from mio_cua.agent_factory import Agent
|
|
10
|
+
return Agent
|
|
11
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
__all__ = ["Agent", "AgentConfig", "Task", "__version__"]
|
|
File without changes
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
from mio_cua.models.observation import Change, ObservationDiff
|
|
2
|
+
from mio_cua.scene.diff import diff as scene_diff_func
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def _key(e):
|
|
6
|
+
return e.role or "unknown", e.text or ""
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _nearby(a, b, tol: int = 5) -> bool:
|
|
10
|
+
"""True if bbox centers are within tol pixels of each other."""
|
|
11
|
+
ax, ay, aw, ah = a
|
|
12
|
+
bx, by, bw, bh = b
|
|
13
|
+
ac = (ax + aw / 2, ay + ah / 2)
|
|
14
|
+
bc = (bx + bw / 2, by + bh / 2)
|
|
15
|
+
return abs(ac[0] - bc[0]) <= tol and abs(ac[1] - bc[1]) <= tol
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _role(e):
|
|
19
|
+
return e.role or "unknown"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def compute_diff(prev, current) -> ObservationDiff:
|
|
23
|
+
if prev is None:
|
|
24
|
+
return ObservationDiff(prev=None, current=current, changes=[])
|
|
25
|
+
# Prefer the scene graph diff when both observations carry one: it detects
|
|
26
|
+
# display readout changes (e.g. calculator 0 -> 7) which the element-level
|
|
27
|
+
# diff treats as unrelated "removed" + "added" noise.
|
|
28
|
+
prev_scene = getattr(prev, "scene", None)
|
|
29
|
+
curr_scene = getattr(current, "scene", None)
|
|
30
|
+
if prev_scene is not None and curr_scene is not None \
|
|
31
|
+
and getattr(prev_scene, "nodes", None) and getattr(curr_scene, "nodes", None):
|
|
32
|
+
sc = scene_diff_func(prev_scene, curr_scene)
|
|
33
|
+
if sc:
|
|
34
|
+
return ObservationDiff(prev=prev, current=current, changes=[
|
|
35
|
+
Change(c.kind, c.node_id, c.description) for c in sc
|
|
36
|
+
])
|
|
37
|
+
return _diff_elements(prev, current)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _diff_elements(prev, current) -> ObservationDiff:
|
|
41
|
+
prev_elements = list(prev.elements)
|
|
42
|
+
changes = []
|
|
43
|
+
matched_prev = set()
|
|
44
|
+
consumed_cur = set()
|
|
45
|
+
|
|
46
|
+
# stable: current element matched to prev by (role, text) identity within bbox tolerance
|
|
47
|
+
for ci, e in enumerate(current.elements):
|
|
48
|
+
for i, p in enumerate(prev_elements):
|
|
49
|
+
if i in matched_prev:
|
|
50
|
+
continue
|
|
51
|
+
if _key(p) == _key(e) and _nearby(p.bbox, e.bbox):
|
|
52
|
+
matched_prev.add(i)
|
|
53
|
+
consumed_cur.add(ci)
|
|
54
|
+
break
|
|
55
|
+
|
|
56
|
+
# text_changed: same role at nearly same position, but text differs
|
|
57
|
+
for ci, e in enumerate(current.elements):
|
|
58
|
+
if ci in consumed_cur:
|
|
59
|
+
continue
|
|
60
|
+
for i, p in enumerate(prev_elements):
|
|
61
|
+
if i in matched_prev:
|
|
62
|
+
continue
|
|
63
|
+
if _role(p) == _role(e) and _nearby(p.bbox, e.bbox) and p.text != e.text:
|
|
64
|
+
matched_prev.add(i)
|
|
65
|
+
consumed_cur.add(ci)
|
|
66
|
+
changes.append(Change("text_changed", e.id, f"{p.text} -> {e.text}"))
|
|
67
|
+
break
|
|
68
|
+
|
|
69
|
+
# added: current elements never matched
|
|
70
|
+
for ci, e in enumerate(current.elements):
|
|
71
|
+
if ci not in consumed_cur:
|
|
72
|
+
changes.append(Change("added", e.id, e.text or e.role))
|
|
73
|
+
|
|
74
|
+
# removed: prev elements never matched
|
|
75
|
+
for i, p in enumerate(prev_elements):
|
|
76
|
+
if i not in matched_prev:
|
|
77
|
+
label = p.text or p.role or "unknown"
|
|
78
|
+
changes.append(Change("removed", None, label))
|
|
79
|
+
|
|
80
|
+
return ObservationDiff(prev=prev, current=current, changes=changes)
|
|
81
|
+
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
from typing import Callable, List
|
|
2
|
+
|
|
3
|
+
from mio_cua.models.action import Plan
|
|
4
|
+
from mio_cua.models.action_result import ActionResult
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class Dispatcher:
|
|
8
|
+
"""Executes a plan of actions through a tool registry.
|
|
9
|
+
|
|
10
|
+
The `recover` callable must accept `(action, result, ctx)` and return an
|
|
11
|
+
ActionResult. If None, a default recovery that re-dispatches the action
|
|
12
|
+
through the registry (bound to ctx) is used.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
def __init__(self, registry, recover: Callable = None):
|
|
16
|
+
self.registry = registry
|
|
17
|
+
self.recover = recover
|
|
18
|
+
|
|
19
|
+
def _default_recover(self, action, result, ctx):
|
|
20
|
+
return result
|
|
21
|
+
|
|
22
|
+
def execute(self, plan: Plan, safety, ctx) -> List[ActionResult]:
|
|
23
|
+
results = []
|
|
24
|
+
for action in plan.actions:
|
|
25
|
+
if safety.should_stop():
|
|
26
|
+
break
|
|
27
|
+
ctx.current_action_id = action.id
|
|
28
|
+
try:
|
|
29
|
+
result = self.registry.call(action.type, action.params, ctx)
|
|
30
|
+
except Exception as e:
|
|
31
|
+
result = ActionResult(action.id, success=False, message=str(e), retryable=True)
|
|
32
|
+
if not result.success and result.retryable:
|
|
33
|
+
if self.recover is not None:
|
|
34
|
+
result = self.recover(action, result, ctx)
|
|
35
|
+
else:
|
|
36
|
+
result = self._default_recover(action, result, ctx)
|
|
37
|
+
results.append(result)
|
|
38
|
+
safety.record_step()
|
|
39
|
+
return results
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""Programmatic action verification against an affordance's ``expected``.
|
|
2
|
+
|
|
3
|
+
The perception layer annotates click candidates with an ``expected`` dict
|
|
4
|
+
(e.g. ``{'display': True}`` for calculator digits, ``{'display': 'unchanged'}``
|
|
5
|
+
for operators). The LLM sees this hint but the loop never checks whether the
|
|
6
|
+
screen actually changed as expected. This module closes that gap: after an
|
|
7
|
+
action, diff the display nodes and report whether the expected change happened,
|
|
8
|
+
so the loop can tell the agent "that click registered" vs "it did not".
|
|
9
|
+
|
|
10
|
+
This reduces the classic failure where the agent clicks a key, the display did
|
|
11
|
+
not change, and it either repeats the click (thinking it missed) or moves on
|
|
12
|
+
(not noticing it missed).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from typing import Optional, Tuple
|
|
16
|
+
|
|
17
|
+
from mio_cua.scene.diff import display_text
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class ExpectedVerifier:
|
|
21
|
+
"""Verify an action's outcome against its expected screen change."""
|
|
22
|
+
|
|
23
|
+
def verify(self, prev_scene, curr_scene, expected: dict) -> Tuple[bool, str]:
|
|
24
|
+
"""Return (ok, detail).
|
|
25
|
+
|
|
26
|
+
``ok`` is True when the display change matches ``expected`` (or there
|
|
27
|
+
is nothing to verify). ``detail`` is a human-readable reason.
|
|
28
|
+
"""
|
|
29
|
+
if not expected:
|
|
30
|
+
return True, "no expectation"
|
|
31
|
+
if "display" not in expected:
|
|
32
|
+
return True, "no display expectation"
|
|
33
|
+
return self._verify_display(prev_scene, curr_scene, expected["display"])
|
|
34
|
+
|
|
35
|
+
def _verify_display(self, prev_scene, curr_scene, want) -> Tuple[bool, str]:
|
|
36
|
+
prev_text = display_text(prev_scene) if prev_scene is not None else ""
|
|
37
|
+
curr_text = display_text(curr_scene) if curr_scene is not None else ""
|
|
38
|
+
changed = bool(prev_text and prev_text != curr_text)
|
|
39
|
+
|
|
40
|
+
if want is True:
|
|
41
|
+
if changed:
|
|
42
|
+
return True, f"display changed: {prev_text!r} -> {curr_text!r}"
|
|
43
|
+
return False, f"display did not change (still {curr_text!r})"
|
|
44
|
+
if want == "unchanged":
|
|
45
|
+
if changed:
|
|
46
|
+
return False, f"display changed unexpectedly: {prev_text!r} -> {curr_text!r}"
|
|
47
|
+
return True, f"display unchanged ({curr_text!r})"
|
|
48
|
+
return True, "unknown expectation"
|
|
@@ -0,0 +1,358 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import os
|
|
3
|
+
import time
|
|
4
|
+
import uuid
|
|
5
|
+
|
|
6
|
+
from mio_cua.agent.diff import compute_diff
|
|
7
|
+
from mio_cua.automation.input_controller import InputController
|
|
8
|
+
from mio_cua.events import ObservationCreated, ActionStarted, ActionFinished, TaskFinished
|
|
9
|
+
from mio_cua.models.action_result import ActionResult
|
|
10
|
+
from mio_cua.models.task import Task, TaskResult
|
|
11
|
+
from mio_cua.scene.memory import SceneMemory
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _keys_eq(sig, want):
|
|
17
|
+
"""True if a key() action sig's keys value is exactly ``want``.
|
|
18
|
+
|
|
19
|
+
The sig looks like ``key([('keys', 'ctrl+s')])`` (real loop) or
|
|
20
|
+
``key({'keys': 'ctrl+s'})`` (older/test). Compare the exact token so
|
|
21
|
+
``ctrl+shift+n`` is not mistaken for ``ctrl+s`` (substring trap).
|
|
22
|
+
"""
|
|
23
|
+
for sep in ("'keys', '", "'keys': '"):
|
|
24
|
+
marker = "'keys'" + sep[6:]
|
|
25
|
+
idx = sig.find(sep)
|
|
26
|
+
if idx < 0:
|
|
27
|
+
continue
|
|
28
|
+
start = idx + len(sep)
|
|
29
|
+
end = sig.find("'", start)
|
|
30
|
+
if end < 0:
|
|
31
|
+
continue
|
|
32
|
+
if sig[start:end] == want:
|
|
33
|
+
return True
|
|
34
|
+
return False
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class AgentLoop:
|
|
38
|
+
def __init__(self, perception, planner, registry, safety, events,
|
|
39
|
+
recover=None, config=None, history=None, controller=None,
|
|
40
|
+
artifact_store=None, state_dir=None):
|
|
41
|
+
self.perception = perception
|
|
42
|
+
self.planner = planner
|
|
43
|
+
self.registry = registry
|
|
44
|
+
self.safety = safety
|
|
45
|
+
self.events = events
|
|
46
|
+
self.recover = recover
|
|
47
|
+
self.config = config
|
|
48
|
+
self.history = history
|
|
49
|
+
self.controller = controller or InputController()
|
|
50
|
+
self.artifact_store = artifact_store
|
|
51
|
+
self.state_dir = state_dir
|
|
52
|
+
self._task = None
|
|
53
|
+
self._artifact_paths = []
|
|
54
|
+
self._task_id = uuid.uuid4().hex[:8]
|
|
55
|
+
self.scene_memory = SceneMemory()
|
|
56
|
+
|
|
57
|
+
def _make_ctx(self, obs):
|
|
58
|
+
from mio_cua.tools.context import ToolContext
|
|
59
|
+
self.controller.current_observation = obs
|
|
60
|
+
return ToolContext(
|
|
61
|
+
controller=self.controller,
|
|
62
|
+
perception=self.perception,
|
|
63
|
+
config=self.config,
|
|
64
|
+
events=self.events,
|
|
65
|
+
current_observation=obs,
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
def _save_artifact(self, obs, action, result):
|
|
69
|
+
if self.artifact_store is None:
|
|
70
|
+
return
|
|
71
|
+
p = self.artifact_store.save_artifact(obs=obs, action=action, result=result,
|
|
72
|
+
task_id=self._task_id)
|
|
73
|
+
self._artifact_paths.append(str(p))
|
|
74
|
+
|
|
75
|
+
def _save_state(self, obs, step):
|
|
76
|
+
if self.state_dir is None:
|
|
77
|
+
return
|
|
78
|
+
from mio_cua.memory.state import TaskState, state_path
|
|
79
|
+
shot = obs.screenshot_path if obs else ""
|
|
80
|
+
TaskState(state_path(self.state_dir, self._task_id)).save(
|
|
81
|
+
task_id=self._task_id,
|
|
82
|
+
instruction=self._task.instruction if self._task else "",
|
|
83
|
+
step=step,
|
|
84
|
+
screenshot=shot,
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
def _prune_artifacts(self):
|
|
88
|
+
if self.artifact_store is None:
|
|
89
|
+
return
|
|
90
|
+
limit = getattr(self.config, "artifact_max_bytes", 200 * 1024 * 1024)
|
|
91
|
+
try:
|
|
92
|
+
freed = self.artifact_store.prune(limit)
|
|
93
|
+
if freed:
|
|
94
|
+
logger.info("pruned %d bytes of old artifacts", freed)
|
|
95
|
+
except Exception as e:
|
|
96
|
+
logger.warning("artifact prune failed: %s", e)
|
|
97
|
+
|
|
98
|
+
def run(self, task: Task) -> TaskResult:
|
|
99
|
+
start = time.time()
|
|
100
|
+
self._task = task
|
|
101
|
+
self.safety.start()
|
|
102
|
+
steps = 0
|
|
103
|
+
finished_status = None
|
|
104
|
+
finished_summary = ""
|
|
105
|
+
terminal = "RUNNING"
|
|
106
|
+
try:
|
|
107
|
+
from collections import deque
|
|
108
|
+
from mio_cua.agent.expected import ExpectedVerifier
|
|
109
|
+
prev = None
|
|
110
|
+
no_change = 0
|
|
111
|
+
repeat_count = 0
|
|
112
|
+
self._recent_sigs = deque(maxlen=8)
|
|
113
|
+
self._verifier = ExpectedVerifier()
|
|
114
|
+
self._pending_verify = None # (node_id, expected, prev_scene) awaiting the next observation
|
|
115
|
+
while not self.safety.should_stop():
|
|
116
|
+
obs = self.perception.observe()
|
|
117
|
+
self.events.publish(ObservationCreated(obs))
|
|
118
|
+
self._save_state(obs, steps)
|
|
119
|
+
self.scene_memory.push(getattr(obs, "scene", None))
|
|
120
|
+
diff = compute_diff(prev, obs)
|
|
121
|
+
if prev is not None and not diff.changes:
|
|
122
|
+
no_change += 1
|
|
123
|
+
else:
|
|
124
|
+
no_change = 0
|
|
125
|
+
hints = []
|
|
126
|
+
if self._pending_verify is not None:
|
|
127
|
+
vh = self._verify_pending(obs)
|
|
128
|
+
if vh:
|
|
129
|
+
hints.append(vh)
|
|
130
|
+
self._pending_verify = None
|
|
131
|
+
mem_summary = self.scene_memory.summarize(
|
|
132
|
+
recent_actions=[h["type"] for h in (self.history.recent(6) if self.history else [])]
|
|
133
|
+
if self.history else None,
|
|
134
|
+
)
|
|
135
|
+
if mem_summary:
|
|
136
|
+
hints.append("MEMORY (what you have already seen/done):\n" + mem_summary +
|
|
137
|
+
"\nUse this to continue the task -- do not re-read or re-open what you already saw.")
|
|
138
|
+
if no_change >= 2:
|
|
139
|
+
hints.append('the screen did not change after your recent actions — the last action had no visible effect. Do NOT repeat it. To confirm a dialog, call key(keys="enter") or click the Save/OK button.')
|
|
140
|
+
confirm_hint = self._confirm_hint()
|
|
141
|
+
if confirm_hint:
|
|
142
|
+
hints.append(confirm_hint)
|
|
143
|
+
rename_hint = self._rename_hint()
|
|
144
|
+
if rename_hint:
|
|
145
|
+
hints.append(rename_hint)
|
|
146
|
+
finish_hint = self._completion_hint(no_change)
|
|
147
|
+
if finish_hint:
|
|
148
|
+
hints.append(finish_hint)
|
|
149
|
+
if len(self._recent_sigs) >= 4 and self._recent_sigs.count(self._recent_sigs[-1]) >= 4:
|
|
150
|
+
hints.append(f"you have called `{self._recent_sigs[-1]}` repeatedly with no effect. STOP repeating it and choose a different action now.")
|
|
151
|
+
if hints:
|
|
152
|
+
logger.debug("hints@%d: %s", steps, " | ".join(hints))
|
|
153
|
+
plan = self.planner.plan(task, obs, diff, self.registry.schemas(), history=self.history, hints=hints)
|
|
154
|
+
if not plan.actions:
|
|
155
|
+
break
|
|
156
|
+
ctx = self._make_ctx(obs)
|
|
157
|
+
for i, action in enumerate(plan.actions):
|
|
158
|
+
if self.safety.should_stop():
|
|
159
|
+
break
|
|
160
|
+
self.events.publish(ActionStarted(action))
|
|
161
|
+
ctx.current_action_id = action.id
|
|
162
|
+
try:
|
|
163
|
+
result = self.registry.call(action.type, action.params, ctx)
|
|
164
|
+
except Exception as e:
|
|
165
|
+
result = ActionResult(action.id, success=False, message=str(e), retryable=True)
|
|
166
|
+
if not result.success and result.retryable and self.recover is not None:
|
|
167
|
+
result = self.recover(action, result, ctx)
|
|
168
|
+
if result.success and action.type == "click":
|
|
169
|
+
self._pending_verify = self._capture_expected(obs, action)
|
|
170
|
+
self._save_artifact(obs, action, result)
|
|
171
|
+
self.events.publish(ActionFinished(result))
|
|
172
|
+
if self.history is not None:
|
|
173
|
+
self.history.record(action.id, action.type, result.success, result.message)
|
|
174
|
+
self.safety.record_step()
|
|
175
|
+
steps += 1
|
|
176
|
+
if action.type not in ("success", "fail"):
|
|
177
|
+
sig = f"{action.type}({sorted(action.params.items())})"
|
|
178
|
+
if self._recent_sigs and self._recent_sigs[-1] == sig:
|
|
179
|
+
repeat_count += 1
|
|
180
|
+
else:
|
|
181
|
+
repeat_count = 1
|
|
182
|
+
self._recent_sigs.append(sig)
|
|
183
|
+
if repeat_count >= 6:
|
|
184
|
+
finished_status = "FAIL"
|
|
185
|
+
finished_summary = f"stuck: repeated {sig} {repeat_count} times with no effect"
|
|
186
|
+
break
|
|
187
|
+
if action.type == "success":
|
|
188
|
+
blocker = self._unconfirmed_edit()
|
|
189
|
+
if blocker:
|
|
190
|
+
# hard guard: an element_id-less type (rename box /
|
|
191
|
+
# filename field) was not confirmed with Enter, so a
|
|
192
|
+
# success() would claim an edit that never applied.
|
|
193
|
+
hints.append(blocker)
|
|
194
|
+
self._save_artifact(obs, action, result)
|
|
195
|
+
self.events.publish(ActionFinished(ActionResult(
|
|
196
|
+
action.id, False, blocker, retryable=True)))
|
|
197
|
+
if self.history is not None:
|
|
198
|
+
self.history.record(action.id, action.type, False, blocker)
|
|
199
|
+
self.safety.record_step()
|
|
200
|
+
steps += 1
|
|
201
|
+
continue
|
|
202
|
+
finished_status = "SUCCESS"
|
|
203
|
+
finished_summary = str(action.params.get("result", ""))
|
|
204
|
+
break
|
|
205
|
+
if action.type == "fail":
|
|
206
|
+
finished_status = "FAIL"
|
|
207
|
+
finished_summary = str(action.params.get("reason", ""))
|
|
208
|
+
break
|
|
209
|
+
# Only ONE action per observation: the screen changes after
|
|
210
|
+
# every action, so subsequent actions in the same plan were
|
|
211
|
+
# decided against a stale scene. Re-observe + replan first.
|
|
212
|
+
# (Terminal success/fail already broke above.)
|
|
213
|
+
if i + 1 < len(plan.actions):
|
|
214
|
+
break
|
|
215
|
+
if finished_status in ("SUCCESS", "FAIL"):
|
|
216
|
+
break
|
|
217
|
+
prev = obs
|
|
218
|
+
except Exception as e:
|
|
219
|
+
finished_status = "FAIL"
|
|
220
|
+
finished_summary = f"loop error: {e}"
|
|
221
|
+
finally:
|
|
222
|
+
# capture status BEFORE stopping so stop() doesn't flip it to ABORTED
|
|
223
|
+
terminal = self.safety.status()
|
|
224
|
+
self.safety.stop()
|
|
225
|
+
|
|
226
|
+
status = finished_status or terminal
|
|
227
|
+
if status == "RUNNING":
|
|
228
|
+
status = "FAIL"
|
|
229
|
+
self._prune_artifacts()
|
|
230
|
+
result = TaskResult(
|
|
231
|
+
status=status,
|
|
232
|
+
summary=finished_summary,
|
|
233
|
+
task_id=self._task_id,
|
|
234
|
+
steps=steps,
|
|
235
|
+
duration=time.time() - start,
|
|
236
|
+
artifacts=self._artifact_paths,
|
|
237
|
+
)
|
|
238
|
+
self.events.publish(TaskFinished(result))
|
|
239
|
+
return result
|
|
240
|
+
|
|
241
|
+
def _capture_expected(self, obs, action):
|
|
242
|
+
"""Record the clicked node's expected screen change for verification."""
|
|
243
|
+
scene = getattr(obs, "scene", None)
|
|
244
|
+
if scene is None:
|
|
245
|
+
return None
|
|
246
|
+
node_id = action.params.get("element_id")
|
|
247
|
+
if node_id is None:
|
|
248
|
+
return None
|
|
249
|
+
aff = scene.affordance_for(int(node_id), "click")
|
|
250
|
+
if aff is None or not aff.expected:
|
|
251
|
+
return None
|
|
252
|
+
return (int(node_id), dict(aff.expected), scene)
|
|
253
|
+
|
|
254
|
+
def _verify_pending(self, obs):
|
|
255
|
+
"""Check whether the previous click produced its expected change."""
|
|
256
|
+
node_id, expected, prev_scene = self._pending_verify
|
|
257
|
+
curr_scene = getattr(obs, "scene", None)
|
|
258
|
+
if curr_scene is None:
|
|
259
|
+
return None
|
|
260
|
+
ok, detail = self._verifier.verify(prev_scene, curr_scene, expected)
|
|
261
|
+
if ok:
|
|
262
|
+
return None
|
|
263
|
+
return (f"VERIFICATION: your last click on node {node_id} did not have the "
|
|
264
|
+
f"expected effect ({detail}). It likely missed or the target changed. "
|
|
265
|
+
f"Do NOT repeat it blindly -- re-inspect and pick a fresh target.")
|
|
266
|
+
|
|
267
|
+
def _unconfirmed_edit(self):
|
|
268
|
+
"""A rename/filename type (type WITHOUT element_id) that has NOT been
|
|
269
|
+
confirmed with Enter. success() must be blocked then, or the agent
|
|
270
|
+
claims an edit that never applied (folder/file name unchanged)."""
|
|
271
|
+
sigs = list(getattr(self, "_recent_sigs", None) or [])
|
|
272
|
+
typed = [s for s in sigs if s.startswith("type(") and "element_id" not in s]
|
|
273
|
+
if not typed:
|
|
274
|
+
return None
|
|
275
|
+
if any(self._is_confirming_action(s) for s in sigs):
|
|
276
|
+
return None
|
|
277
|
+
return ("BLOCKED: you typed a name but never pressed `enter` to apply it "
|
|
278
|
+
"-- the edit is NOT saved. Call `key(keys=\"enter\")` to confirm, "
|
|
279
|
+
"then call `success`.")
|
|
280
|
+
|
|
281
|
+
def _rename_hint(self):
|
|
282
|
+
"""If the agent pressed ctrl+shift+n (created a new folder) but has not
|
|
283
|
+
typed a name for it, the folder stays '新建文件夹' and the task is not
|
|
284
|
+
done. Tell it to type the name (type WITHOUT element_id -- the rename
|
|
285
|
+
box has focus)."""
|
|
286
|
+
sigs = list(getattr(self, "_recent_sigs", None) or [])
|
|
287
|
+
if not sigs:
|
|
288
|
+
return None
|
|
289
|
+
created = any("ctrl+shift+n" in s for s in sigs[-4:])
|
|
290
|
+
if not created:
|
|
291
|
+
return None
|
|
292
|
+
if any(s.startswith("type(") for s in sigs[-4:]):
|
|
293
|
+
return None
|
|
294
|
+
return ("You created a new folder (ctrl+shift+n) but have NOT typed its "
|
|
295
|
+
"name. Its name is selected and the rename box has focus -- call "
|
|
296
|
+
"`type` WITHOUT element_id (e.g. `type(text=\"smoke_demo_folder\")`) "
|
|
297
|
+
"to name it, then press `enter`.")
|
|
298
|
+
|
|
299
|
+
def _confirm_hint(self):
|
|
300
|
+
"""If the agent typed WITHOUT an element_id (a focused rename box or
|
|
301
|
+
filename field -- the explorer/filename case) and has not pressed Enter
|
|
302
|
+
to confirm, the edit is still pending. Tell it to press Enter instead
|
|
303
|
+
of re-typing or moving on. The classic explorer failure: type the
|
|
304
|
+
folder name, never press Enter, call success.
|
|
305
|
+
|
|
306
|
+
Types WITH an element_id (e.g. notepad body) do not need Enter, so we
|
|
307
|
+
only react to element_id-less types to avoid false positives.
|
|
308
|
+
"""
|
|
309
|
+
sigs = list(getattr(self, "_recent_sigs", None) or [])
|
|
310
|
+
if not sigs:
|
|
311
|
+
return None
|
|
312
|
+
typed_without_target = [s for s in sigs
|
|
313
|
+
if s.startswith("type(") and "element_id" not in s]
|
|
314
|
+
if not typed_without_target:
|
|
315
|
+
return None
|
|
316
|
+
# ...and no Enter/Save confirmation after the latest such type
|
|
317
|
+
if any(self._is_confirming_action(s) for s in sigs[-3:]):
|
|
318
|
+
return None
|
|
319
|
+
return ("You typed into a rename/filename box (type without element_id) "
|
|
320
|
+
"but have NOT pressed `enter` to confirm it -- the change is NOT "
|
|
321
|
+
"applied until `enter`. Call `key(keys=\"enter\")` now, then the "
|
|
322
|
+
"task is done.")
|
|
323
|
+
|
|
324
|
+
def _completion_hint(self, no_change):
|
|
325
|
+
"""Hint the agent to finish when it has done the closing steps but
|
|
326
|
+
keeps verifying instead of calling success (task completed but
|
|
327
|
+
status=TIMEOUT).
|
|
328
|
+
|
|
329
|
+
Requires a *confirming* action (Enter / Save) reasonably recent AND the
|
|
330
|
+
screen to have settled (no_change >= 1). Not just any type: typing
|
|
331
|
+
without Enter leaves the edit unconfirmed, so that alone is not
|
|
332
|
+
completion (see _confirm_hint). A confirming action later in the
|
|
333
|
+
trailing history means a save/rename was applied.
|
|
334
|
+
"""
|
|
335
|
+
if no_change < 1:
|
|
336
|
+
return None
|
|
337
|
+
sigs = list(getattr(self, "_recent_sigs", None) or [])
|
|
338
|
+
if len(sigs) < 2:
|
|
339
|
+
return None
|
|
340
|
+
confirmed = [s for s in sigs if self._is_confirming_action(s)]
|
|
341
|
+
if not confirmed:
|
|
342
|
+
return None
|
|
343
|
+
# The confirming action should not be buried too far behind unrelated work.
|
|
344
|
+
last_index = max(i for i, s in enumerate(sigs) if self._is_confirming_action(s))
|
|
345
|
+
if len(sigs) - 1 - last_index > 2:
|
|
346
|
+
return None
|
|
347
|
+
return ("The screen has settled and you already pressed Enter / Save to "
|
|
348
|
+
"apply the change. If the task's goal is met, STOP and call "
|
|
349
|
+
"`success` with a summary now -- do not keep acting.")
|
|
350
|
+
|
|
351
|
+
@staticmethod
|
|
352
|
+
def _is_confirming_action(sig):
|
|
353
|
+
"""Actions that CONFIRM a save/rename: Enter, Save click, Ctrl+S."""
|
|
354
|
+
if sig.startswith("key("):
|
|
355
|
+
return any(_keys_eq(sig, k) for k in ("enter", "ctrl+s"))
|
|
356
|
+
if sig.startswith("click("):
|
|
357
|
+
return "Save" in sig or "保存" in sig
|
|
358
|
+
return False
|