mio-cua 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. mio_cua-0.1.0/PKG-INFO +13 -0
  2. mio_cua-0.1.0/README.md +110 -0
  3. mio_cua-0.1.0/mio_cua/__init__.py +14 -0
  4. mio_cua-0.1.0/mio_cua/agent/__init__.py +0 -0
  5. mio_cua-0.1.0/mio_cua/agent/diff.py +81 -0
  6. mio_cua-0.1.0/mio_cua/agent/dispatcher.py +39 -0
  7. mio_cua-0.1.0/mio_cua/agent/expected.py +48 -0
  8. mio_cua-0.1.0/mio_cua/agent/loop.py +358 -0
  9. mio_cua-0.1.0/mio_cua/agent/planner.py +176 -0
  10. mio_cua-0.1.0/mio_cua/agent/recover.py +52 -0
  11. mio_cua-0.1.0/mio_cua/agent/safety.py +68 -0
  12. mio_cua-0.1.0/mio_cua/agent/verify.py +23 -0
  13. mio_cua-0.1.0/mio_cua/agent_factory.py +72 -0
  14. mio_cua-0.1.0/mio_cua/automation/__init__.py +0 -0
  15. mio_cua-0.1.0/mio_cua/automation/backends.py +161 -0
  16. mio_cua-0.1.0/mio_cua/automation/input_controller.py +48 -0
  17. mio_cua-0.1.0/mio_cua/automation/uia.py +48 -0
  18. mio_cua-0.1.0/mio_cua/automation/windows.py +315 -0
  19. mio_cua-0.1.0/mio_cua/cli.py +179 -0
  20. mio_cua-0.1.0/mio_cua/config.py +37 -0
  21. mio_cua-0.1.0/mio_cua/events.py +38 -0
  22. mio_cua-0.1.0/mio_cua/llm/__init__.py +0 -0
  23. mio_cua-0.1.0/mio_cua/llm/client.py +23 -0
  24. mio_cua-0.1.0/mio_cua/mcp_server.py +174 -0
  25. mio_cua-0.1.0/mio_cua/memory/__init__.py +5 -0
  26. mio_cua-0.1.0/mio_cua/memory/artifact.py +96 -0
  27. mio_cua-0.1.0/mio_cua/memory/history.py +12 -0
  28. mio_cua-0.1.0/mio_cua/memory/state.py +23 -0
  29. mio_cua-0.1.0/mio_cua/models/__init__.py +11 -0
  30. mio_cua-0.1.0/mio_cua/models/action.py +23 -0
  31. mio_cua-0.1.0/mio_cua/models/action_result.py +18 -0
  32. mio_cua-0.1.0/mio_cua/models/element.py +16 -0
  33. mio_cua-0.1.0/mio_cua/models/observation.py +27 -0
  34. mio_cua-0.1.0/mio_cua/models/task.py +21 -0
  35. mio_cua-0.1.0/mio_cua/perception/__init__.py +4 -0
  36. mio_cua-0.1.0/mio_cua/perception/merger.py +77 -0
  37. mio_cua-0.1.0/mio_cua/perception/perception.py +146 -0
  38. mio_cua-0.1.0/mio_cua/prompts/__init__.py +11 -0
  39. mio_cua-0.1.0/mio_cua/prompts/system.txt +35 -0
  40. mio_cua-0.1.0/mio_cua/providers/__init__.py +0 -0
  41. mio_cua-0.1.0/mio_cua/providers/base.py +15 -0
  42. mio_cua-0.1.0/mio_cua/providers/openai_compat.py +38 -0
  43. mio_cua-0.1.0/mio_cua/scene/__init__.py +60 -0
  44. mio_cua-0.1.0/mio_cua/scene/affordances.py +221 -0
  45. mio_cua-0.1.0/mio_cua/scene/builder.py +140 -0
  46. mio_cua-0.1.0/mio_cua/scene/diff.py +86 -0
  47. mio_cua-0.1.0/mio_cua/scene/graph.py +102 -0
  48. mio_cua-0.1.0/mio_cua/scene/memory.py +67 -0
  49. mio_cua-0.1.0/mio_cua/scene/omniparser.py +127 -0
  50. mio_cua-0.1.0/mio_cua/scene/regions.py +89 -0
  51. mio_cua-0.1.0/mio_cua/scene/relations.py +109 -0
  52. mio_cua-0.1.0/mio_cua/simulation.py +272 -0
  53. mio_cua-0.1.0/mio_cua/tools/__init__.py +0 -0
  54. mio_cua-0.1.0/mio_cua/tools/builtin.py +58 -0
  55. mio_cua-0.1.0/mio_cua/tools/click.py +14 -0
  56. mio_cua-0.1.0/mio_cua/tools/context.py +14 -0
  57. mio_cua-0.1.0/mio_cua/tools/fail.py +7 -0
  58. mio_cua-0.1.0/mio_cua/tools/focus_window.py +10 -0
  59. mio_cua-0.1.0/mio_cua/tools/fs.py +88 -0
  60. mio_cua-0.1.0/mio_cua/tools/key.py +7 -0
  61. mio_cua-0.1.0/mio_cua/tools/launch.py +111 -0
  62. mio_cua-0.1.0/mio_cua/tools/move_mouse.py +14 -0
  63. mio_cua-0.1.0/mio_cua/tools/registry.py +21 -0
  64. mio_cua-0.1.0/mio_cua/tools/screenshot.py +18 -0
  65. mio_cua-0.1.0/mio_cua/tools/scroll.py +7 -0
  66. mio_cua-0.1.0/mio_cua/tools/success.py +7 -0
  67. mio_cua-0.1.0/mio_cua/tools/type.py +27 -0
  68. mio_cua-0.1.0/mio_cua/tools/wait.py +8 -0
  69. mio_cua-0.1.0/mio_cua/vision/__init__.py +0 -0
  70. mio_cua-0.1.0/mio_cua/vision/ocr.py +47 -0
  71. mio_cua-0.1.0/mio_cua/vision/overlay.py +13 -0
  72. mio_cua-0.1.0/mio_cua/vision/screen.py +39 -0
  73. mio_cua-0.1.0/mio_cua.egg-info/PKG-INFO +13 -0
  74. mio_cua-0.1.0/mio_cua.egg-info/SOURCES.txt +78 -0
  75. mio_cua-0.1.0/mio_cua.egg-info/dependency_links.txt +1 -0
  76. mio_cua-0.1.0/mio_cua.egg-info/entry_points.txt +3 -0
  77. mio_cua-0.1.0/mio_cua.egg-info/requires.txt +8 -0
  78. mio_cua-0.1.0/mio_cua.egg-info/top_level.txt +1 -0
  79. mio_cua-0.1.0/pyproject.toml +33 -0
  80. mio_cua-0.1.0/setup.cfg +4 -0
mio_cua-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,13 @@
1
+ Metadata-Version: 2.4
2
+ Name: mio-cua
3
+ Version: 0.1.0
4
+ Summary: Mio Computer-Use Agent: Windows desktop automation AI agent (SDK + CLI)
5
+ Requires-Python: >=3.10
6
+ Requires-Dist: requests>=2.31
7
+ Requires-Dist: mss>=9.0
8
+ Requires-Dist: Pillow>=10.0
9
+ Requires-Dist: PyYAML>=6.0
10
+ Requires-Dist: pynput>=1.7
11
+ Requires-Dist: pywin32>=306
12
+ Requires-Dist: pywinauto>=0.6.8
13
+ Requires-Dist: rapidocr_onnxruntime>=1.3.24
@@ -0,0 +1,110 @@
1
+ # mio-cua
2
+
3
+ Mio Computer-Use Agent:Windows 桌面自动化 AI Agent(Python SDK + CLI)。
4
+
5
+ <!-- mcp-name: io.github.mldlbs/mio-cua -->
6
+
7
+ ## 安装
8
+
9
+ ```bash
10
+ pip install -e . # 从源码安装(依赖见 pyproject.toml)
11
+ # Windows 下建议以管理员身份运行
12
+ ```
13
+
14
+ ### 配置
15
+
16
+ 设置 LLM API Key:
17
+
18
+ ```bash
19
+ set OPENAI_API_KEY=sk-xxx # Windows CMD
20
+ $env:OPENAI_API_KEY = "sk-xxx" # PowerShell
21
+ ```
22
+
23
+ 示例脚本还支持 `DESKTOP_AGENT_PROVIDER` / `DESKTOP_AGENT_MODEL` 环境变量覆盖模型。
24
+
25
+ ### 当前状态
26
+
27
+ v0.5:Scene Graph 感知层 + 网页纯视觉感知(Regions 版面 + OmniParser 控件识别);五个端到端场景(记事本/计算器/资源管理器/跨应用/网页)已在真实 Windows 11 桌面完成 5/5 全绿终验(2026-08-08)。
28
+
29
+ - **Scene Graph 感知层**:感知层把 OCR+UIA 融合为一张场景图(`mio_cua/scene/`)——每个 UI 对象是一个 Node(类型/文本/状态/bbox),配空间关系(leftOf/above/labelFor)和 **Action candidates**(感知层已验证的可点击动作,如 `click node 7 ('7') {'value': '7'} expects {'display': True}`)。LLM 从候选动作里选择,不再自行推断坐标/UI。
30
+ - **网页纯视觉感知(像人一样上网)**:浏览器窗口额外走两条视觉链路,不依赖 DOM——
31
+ - **Regions 版面分析**(`rapid_layout`,可选依赖):识别页面结构(导航/标题/正文/表格/图片区域),可选依赖未装则优雅降级。
32
+ - **OmniParser 控件识别**(`scene/omniparser.py`,可选依赖):YOLOv9 交互元素检测 + Florence-2 语义描述,把网页截图解析成按钮/链接/输入框(`button`/`text` 节点 + 可点击候选)。需 torch CUDA;模型默认在项目内 `models/omniparser/`,也可用 `OMNIPARSER_DIR`/`OMNIPARSER_WEIGHTS` 环境变量覆盖。
33
+ - **显示区验证**:计算器显示区被识别为 `display` 节点,点击数字键后通过 Scene Diff 检测显示值变化(`0`→`7`),用于动作成功校验。
34
+ - **跨应用(crossapp)场景**:`smoke/crossapp.yaml`——读 `smoke_numbers.txt`(12/34/56)→ 计算器求和 102 → 保存 `smoke_sum_result.txt`。全绿验收(2026-08-08,19-21 步)。关键修复:**一动作一感知**(每步重新感知再决策,避免多动作基于陈旧场景执行);**计算器 `+` 键发送**(`key(keys="+")` 曾被当分隔符 split 失败);**UWP 计算器聚焦**;其余见 SMOKE.md 修复链。
35
+ - **网页纯视觉场景(web)**:`smoke/web.yaml`——Edge 打开本地 HTML,纯视觉(OmniParser,不接 DOM)点击按钮/输入文本并视觉确认。PASS(2026-08-08,9 步 165s)。OmniParser 走本地 HF 缓存离线推理(`HF_HUB_OFFLINE=1`,首帧加载约 20s,后续每帧 ~0.3s)。
36
+ - **完整套件**:calculator/crossapp/explorer/notepad/web 五场景全绿终验(2026-08-08,crossapp 21 步 SUCCESS、其余 SUCCESS/校验通过)。deepseek-v4-flash 即可完成。调优中修复 6 个工程 bug(陈旧场景执行、UWP 聚焦、`+` 键 split、浏览器 PATH、重命名引导、元素 id 稳定)。
37
+ - **GPU OCR**:经 DirectML(onnxruntime-directml)加速,单次 OCR 平均 ~1.6s;`DESKTOP_AGENT_OCR_DEVICE=cpu` 可回退纯 CPU。
38
+ - **窗口区域截图**:只抓活动窗口区域(而非全屏),截图体积缩减约 80%,更聚焦目标窗口。
39
+ - **Artifact 自动清理**:任务结束后自动修剪,目录默认上限 200MB(配置项 `artifact_max_bytes`)。
40
+ - **循环守卫**:动作无变化 / 重复相同动作 ≥6 次直接判定 FAIL,避免空转。
41
+ - **真实验收(deepseek-v4-flash + DirectML)**:记事本(输入并保存 `hello world`)、计算器(`123*456=56088`)、资源管理器(新建并重命名 `smoke_demo_folder`)三个场景全部 PASS。
42
+
43
+ 发送给 LLM 的截图带编号框(overlay),编号与元素列表 `id` 一一对应。每步操作与截图会落盘为 artifact,任务状态支持 `resume`。可重试失败会自动触发 Recovery(聚焦窗口后重试,每次 action 最多重试 2 次)。
44
+
45
+ > 提示:首次在真实桌面运行前,请先完成一次小任务的冒烟测试(如"打开记事本输入 hello"),并确认 F9 急停可用。冒烟测试见 `SMOKE.md`。
46
+
47
+ ## CLI
48
+
49
+ ```bash
50
+ mio-cua run "打开计算器,计算 3*4" --model gpt-4o
51
+ mio-cua run "删除所有文件" --dry-run # 只打印计划,不执行
52
+ mio-cua resume <task_id> # 从上次状态重新执行任务
53
+ mio-cua replay <task_id> # 从 artifact 回放任务步骤(调试用,--full 显示参数)
54
+ mio-cua providers
55
+ ```
56
+
57
+ ## Artifacts / 状态
58
+
59
+ - 每步:`~/.mio_cua/artifacts/<ts>.json`(observation + action + result)
60
+ - 每步截图:`~/.mio_cua/artifacts/<ts>.png`(overlay 标注图)与 `.raw.png`
61
+ - 任务状态:`~/.mio_cua/artifacts/state/<task_id>.json`
62
+
63
+ ## SDK
64
+
65
+ ```python
66
+ from mio_cua import Agent, AgentConfig, Task
67
+
68
+ agent = Agent(AgentConfig(model="gpt-4o", max_steps=50))
69
+ result = agent.run(Task(instruction="打开记事本,输入 hello"))
70
+ print(result.status, result.steps)
71
+ ```
72
+
73
+ ## MCP(接入 Claude / Cursor / ChatGPT)
74
+
75
+ mio-cua 也作为 **MCP server** 暴露,让任意 MCP 客户端直接控制桌面:
76
+
77
+ ```bash
78
+ pip install -e .
79
+ ```
80
+
81
+ ```json
82
+ { "mcpServers": { "mio-cua": { "command": "mio-cua-mcp", "args": [] } } }
83
+ ```
84
+
85
+ 10 个工具:文件操作(list_dir/make_dir/move_file/move_files)、窗口(launch/focus_window/get_active_window)、输入(click/type/key)。详见 [MCP.md](MCP.md)。
86
+
87
+ ## 安全
88
+
89
+ - F9 急停
90
+ - 步数上限 / 任务超时
91
+ - 每步截图留痕(artifact)
92
+
93
+ ## 验收套件(smoke)
94
+
95
+ 五个端到端场景在**独立虚拟桌面**上隔离运行,不打扰主桌面:
96
+
97
+ ```bash
98
+ # 完整套件(calculator/crossapp/explorer/notepad/web)
99
+ python scripts/run_smoke_vdesk.py --only calculator,crossapp,explorer,notepad,web --model deepseek-v4-flash --base-url https://api.deepseek.com/v1
100
+
101
+ # 单场景
102
+ python scripts/run_smoke_vdesk.py --only crossapp --model deepseek-v4-flash --base-url https://api.deepseek.com/v1
103
+ ```
104
+
105
+ 运行前需:`$env:OPENAI_API_KEY="sk-xxx"`;桌面已解锁;用户约 20 分钟不碰键鼠(vdesk 隔离在用户操作时失效)。日志 `%TEMP%\smoke_vdesk.log`,结果以 `[PASS]/[FAIL]` 汇总。
106
+
107
+ 前置文件:`~/Desktop/smoke_numbers.txt`(12/34/56,crossapp 输入)、`~/Desktop/vision_test.html`(web 测试页)。场景间自动清理测试应用进程。
108
+
109
+ - 判定规则:`status=SUCCESS` 且(若场景有产物校验)所有 file/dir/contains 校验通过才算 PASS——防止 agent 假 success。
110
+ - 收尾提示:屏幕稳定且已按 Enter/Save 确认后,提示 agent 调 `success`,避免任务完成却超时。
@@ -0,0 +1,14 @@
1
+ __version__ = "0.1.0"
2
+
3
+ from mio_cua.config import AgentConfig
4
+ from mio_cua.models.task import Task
5
+
6
+
7
+ def __getattr__(name):
8
+ if name == "Agent":
9
+ from mio_cua.agent_factory import Agent
10
+ return Agent
11
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
12
+
13
+
14
+ __all__ = ["Agent", "AgentConfig", "Task", "__version__"]
File without changes
@@ -0,0 +1,81 @@
1
+ from mio_cua.models.observation import Change, ObservationDiff
2
+ from mio_cua.scene.diff import diff as scene_diff_func
3
+
4
+
5
+ def _key(e):
6
+ return e.role or "unknown", e.text or ""
7
+
8
+
9
+ def _nearby(a, b, tol: int = 5) -> bool:
10
+ """True if bbox centers are within tol pixels of each other."""
11
+ ax, ay, aw, ah = a
12
+ bx, by, bw, bh = b
13
+ ac = (ax + aw / 2, ay + ah / 2)
14
+ bc = (bx + bw / 2, by + bh / 2)
15
+ return abs(ac[0] - bc[0]) <= tol and abs(ac[1] - bc[1]) <= tol
16
+
17
+
18
+ def _role(e):
19
+ return e.role or "unknown"
20
+
21
+
22
+ def compute_diff(prev, current) -> ObservationDiff:
23
+ if prev is None:
24
+ return ObservationDiff(prev=None, current=current, changes=[])
25
+ # Prefer the scene graph diff when both observations carry one: it detects
26
+ # display readout changes (e.g. calculator 0 -> 7) which the element-level
27
+ # diff treats as unrelated "removed" + "added" noise.
28
+ prev_scene = getattr(prev, "scene", None)
29
+ curr_scene = getattr(current, "scene", None)
30
+ if prev_scene is not None and curr_scene is not None \
31
+ and getattr(prev_scene, "nodes", None) and getattr(curr_scene, "nodes", None):
32
+ sc = scene_diff_func(prev_scene, curr_scene)
33
+ if sc:
34
+ return ObservationDiff(prev=prev, current=current, changes=[
35
+ Change(c.kind, c.node_id, c.description) for c in sc
36
+ ])
37
+ return _diff_elements(prev, current)
38
+
39
+
40
+ def _diff_elements(prev, current) -> ObservationDiff:
41
+ prev_elements = list(prev.elements)
42
+ changes = []
43
+ matched_prev = set()
44
+ consumed_cur = set()
45
+
46
+ # stable: current element matched to prev by (role, text) identity within bbox tolerance
47
+ for ci, e in enumerate(current.elements):
48
+ for i, p in enumerate(prev_elements):
49
+ if i in matched_prev:
50
+ continue
51
+ if _key(p) == _key(e) and _nearby(p.bbox, e.bbox):
52
+ matched_prev.add(i)
53
+ consumed_cur.add(ci)
54
+ break
55
+
56
+ # text_changed: same role at nearly same position, but text differs
57
+ for ci, e in enumerate(current.elements):
58
+ if ci in consumed_cur:
59
+ continue
60
+ for i, p in enumerate(prev_elements):
61
+ if i in matched_prev:
62
+ continue
63
+ if _role(p) == _role(e) and _nearby(p.bbox, e.bbox) and p.text != e.text:
64
+ matched_prev.add(i)
65
+ consumed_cur.add(ci)
66
+ changes.append(Change("text_changed", e.id, f"{p.text} -> {e.text}"))
67
+ break
68
+
69
+ # added: current elements never matched
70
+ for ci, e in enumerate(current.elements):
71
+ if ci not in consumed_cur:
72
+ changes.append(Change("added", e.id, e.text or e.role))
73
+
74
+ # removed: prev elements never matched
75
+ for i, p in enumerate(prev_elements):
76
+ if i not in matched_prev:
77
+ label = p.text or p.role or "unknown"
78
+ changes.append(Change("removed", None, label))
79
+
80
+ return ObservationDiff(prev=prev, current=current, changes=changes)
81
+
@@ -0,0 +1,39 @@
1
+ from typing import Callable, List
2
+
3
+ from mio_cua.models.action import Plan
4
+ from mio_cua.models.action_result import ActionResult
5
+
6
+
7
+ class Dispatcher:
8
+ """Executes a plan of actions through a tool registry.
9
+
10
+ The `recover` callable must accept `(action, result, ctx)` and return an
11
+ ActionResult. If None, a default recovery that re-dispatches the action
12
+ through the registry (bound to ctx) is used.
13
+ """
14
+
15
+ def __init__(self, registry, recover: Callable = None):
16
+ self.registry = registry
17
+ self.recover = recover
18
+
19
+ def _default_recover(self, action, result, ctx):
20
+ return result
21
+
22
+ def execute(self, plan: Plan, safety, ctx) -> List[ActionResult]:
23
+ results = []
24
+ for action in plan.actions:
25
+ if safety.should_stop():
26
+ break
27
+ ctx.current_action_id = action.id
28
+ try:
29
+ result = self.registry.call(action.type, action.params, ctx)
30
+ except Exception as e:
31
+ result = ActionResult(action.id, success=False, message=str(e), retryable=True)
32
+ if not result.success and result.retryable:
33
+ if self.recover is not None:
34
+ result = self.recover(action, result, ctx)
35
+ else:
36
+ result = self._default_recover(action, result, ctx)
37
+ results.append(result)
38
+ safety.record_step()
39
+ return results
@@ -0,0 +1,48 @@
1
+ """Programmatic action verification against an affordance's ``expected``.
2
+
3
+ The perception layer annotates click candidates with an ``expected`` dict
4
+ (e.g. ``{'display': True}`` for calculator digits, ``{'display': 'unchanged'}``
5
+ for operators). The LLM sees this hint but the loop never checks whether the
6
+ screen actually changed as expected. This module closes that gap: after an
7
+ action, diff the display nodes and report whether the expected change happened,
8
+ so the loop can tell the agent "that click registered" vs "it did not".
9
+
10
+ This reduces the classic failure where the agent clicks a key, the display did
11
+ not change, and it either repeats the click (thinking it missed) or moves on
12
+ (not noticing it missed).
13
+ """
14
+
15
+ from typing import Optional, Tuple
16
+
17
+ from mio_cua.scene.diff import display_text
18
+
19
+
20
+ class ExpectedVerifier:
21
+ """Verify an action's outcome against its expected screen change."""
22
+
23
+ def verify(self, prev_scene, curr_scene, expected: dict) -> Tuple[bool, str]:
24
+ """Return (ok, detail).
25
+
26
+ ``ok`` is True when the display change matches ``expected`` (or there
27
+ is nothing to verify). ``detail`` is a human-readable reason.
28
+ """
29
+ if not expected:
30
+ return True, "no expectation"
31
+ if "display" not in expected:
32
+ return True, "no display expectation"
33
+ return self._verify_display(prev_scene, curr_scene, expected["display"])
34
+
35
+ def _verify_display(self, prev_scene, curr_scene, want) -> Tuple[bool, str]:
36
+ prev_text = display_text(prev_scene) if prev_scene is not None else ""
37
+ curr_text = display_text(curr_scene) if curr_scene is not None else ""
38
+ changed = bool(prev_text and prev_text != curr_text)
39
+
40
+ if want is True:
41
+ if changed:
42
+ return True, f"display changed: {prev_text!r} -> {curr_text!r}"
43
+ return False, f"display did not change (still {curr_text!r})"
44
+ if want == "unchanged":
45
+ if changed:
46
+ return False, f"display changed unexpectedly: {prev_text!r} -> {curr_text!r}"
47
+ return True, f"display unchanged ({curr_text!r})"
48
+ return True, "unknown expectation"
@@ -0,0 +1,358 @@
1
+ import logging
2
+ import os
3
+ import time
4
+ import uuid
5
+
6
+ from mio_cua.agent.diff import compute_diff
7
+ from mio_cua.automation.input_controller import InputController
8
+ from mio_cua.events import ObservationCreated, ActionStarted, ActionFinished, TaskFinished
9
+ from mio_cua.models.action_result import ActionResult
10
+ from mio_cua.models.task import Task, TaskResult
11
+ from mio_cua.scene.memory import SceneMemory
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+
16
+ def _keys_eq(sig, want):
17
+ """True if a key() action sig's keys value is exactly ``want``.
18
+
19
+ The sig looks like ``key([('keys', 'ctrl+s')])`` (real loop) or
20
+ ``key({'keys': 'ctrl+s'})`` (older/test). Compare the exact token so
21
+ ``ctrl+shift+n`` is not mistaken for ``ctrl+s`` (substring trap).
22
+ """
23
+ for sep in ("'keys', '", "'keys': '"):
24
+ marker = "'keys'" + sep[6:]
25
+ idx = sig.find(sep)
26
+ if idx < 0:
27
+ continue
28
+ start = idx + len(sep)
29
+ end = sig.find("'", start)
30
+ if end < 0:
31
+ continue
32
+ if sig[start:end] == want:
33
+ return True
34
+ return False
35
+
36
+
37
+ class AgentLoop:
38
+ def __init__(self, perception, planner, registry, safety, events,
39
+ recover=None, config=None, history=None, controller=None,
40
+ artifact_store=None, state_dir=None):
41
+ self.perception = perception
42
+ self.planner = planner
43
+ self.registry = registry
44
+ self.safety = safety
45
+ self.events = events
46
+ self.recover = recover
47
+ self.config = config
48
+ self.history = history
49
+ self.controller = controller or InputController()
50
+ self.artifact_store = artifact_store
51
+ self.state_dir = state_dir
52
+ self._task = None
53
+ self._artifact_paths = []
54
+ self._task_id = uuid.uuid4().hex[:8]
55
+ self.scene_memory = SceneMemory()
56
+
57
+ def _make_ctx(self, obs):
58
+ from mio_cua.tools.context import ToolContext
59
+ self.controller.current_observation = obs
60
+ return ToolContext(
61
+ controller=self.controller,
62
+ perception=self.perception,
63
+ config=self.config,
64
+ events=self.events,
65
+ current_observation=obs,
66
+ )
67
+
68
+ def _save_artifact(self, obs, action, result):
69
+ if self.artifact_store is None:
70
+ return
71
+ p = self.artifact_store.save_artifact(obs=obs, action=action, result=result,
72
+ task_id=self._task_id)
73
+ self._artifact_paths.append(str(p))
74
+
75
+ def _save_state(self, obs, step):
76
+ if self.state_dir is None:
77
+ return
78
+ from mio_cua.memory.state import TaskState, state_path
79
+ shot = obs.screenshot_path if obs else ""
80
+ TaskState(state_path(self.state_dir, self._task_id)).save(
81
+ task_id=self._task_id,
82
+ instruction=self._task.instruction if self._task else "",
83
+ step=step,
84
+ screenshot=shot,
85
+ )
86
+
87
+ def _prune_artifacts(self):
88
+ if self.artifact_store is None:
89
+ return
90
+ limit = getattr(self.config, "artifact_max_bytes", 200 * 1024 * 1024)
91
+ try:
92
+ freed = self.artifact_store.prune(limit)
93
+ if freed:
94
+ logger.info("pruned %d bytes of old artifacts", freed)
95
+ except Exception as e:
96
+ logger.warning("artifact prune failed: %s", e)
97
+
98
+ def run(self, task: Task) -> TaskResult:
99
+ start = time.time()
100
+ self._task = task
101
+ self.safety.start()
102
+ steps = 0
103
+ finished_status = None
104
+ finished_summary = ""
105
+ terminal = "RUNNING"
106
+ try:
107
+ from collections import deque
108
+ from mio_cua.agent.expected import ExpectedVerifier
109
+ prev = None
110
+ no_change = 0
111
+ repeat_count = 0
112
+ self._recent_sigs = deque(maxlen=8)
113
+ self._verifier = ExpectedVerifier()
114
+ self._pending_verify = None # (node_id, expected, prev_scene) awaiting the next observation
115
+ while not self.safety.should_stop():
116
+ obs = self.perception.observe()
117
+ self.events.publish(ObservationCreated(obs))
118
+ self._save_state(obs, steps)
119
+ self.scene_memory.push(getattr(obs, "scene", None))
120
+ diff = compute_diff(prev, obs)
121
+ if prev is not None and not diff.changes:
122
+ no_change += 1
123
+ else:
124
+ no_change = 0
125
+ hints = []
126
+ if self._pending_verify is not None:
127
+ vh = self._verify_pending(obs)
128
+ if vh:
129
+ hints.append(vh)
130
+ self._pending_verify = None
131
+ mem_summary = self.scene_memory.summarize(
132
+ recent_actions=[h["type"] for h in (self.history.recent(6) if self.history else [])]
133
+ if self.history else None,
134
+ )
135
+ if mem_summary:
136
+ hints.append("MEMORY (what you have already seen/done):\n" + mem_summary +
137
+ "\nUse this to continue the task -- do not re-read or re-open what you already saw.")
138
+ if no_change >= 2:
139
+ hints.append('the screen did not change after your recent actions — the last action had no visible effect. Do NOT repeat it. To confirm a dialog, call key(keys="enter") or click the Save/OK button.')
140
+ confirm_hint = self._confirm_hint()
141
+ if confirm_hint:
142
+ hints.append(confirm_hint)
143
+ rename_hint = self._rename_hint()
144
+ if rename_hint:
145
+ hints.append(rename_hint)
146
+ finish_hint = self._completion_hint(no_change)
147
+ if finish_hint:
148
+ hints.append(finish_hint)
149
+ if len(self._recent_sigs) >= 4 and self._recent_sigs.count(self._recent_sigs[-1]) >= 4:
150
+ hints.append(f"you have called `{self._recent_sigs[-1]}` repeatedly with no effect. STOP repeating it and choose a different action now.")
151
+ if hints:
152
+ logger.debug("hints@%d: %s", steps, " | ".join(hints))
153
+ plan = self.planner.plan(task, obs, diff, self.registry.schemas(), history=self.history, hints=hints)
154
+ if not plan.actions:
155
+ break
156
+ ctx = self._make_ctx(obs)
157
+ for i, action in enumerate(plan.actions):
158
+ if self.safety.should_stop():
159
+ break
160
+ self.events.publish(ActionStarted(action))
161
+ ctx.current_action_id = action.id
162
+ try:
163
+ result = self.registry.call(action.type, action.params, ctx)
164
+ except Exception as e:
165
+ result = ActionResult(action.id, success=False, message=str(e), retryable=True)
166
+ if not result.success and result.retryable and self.recover is not None:
167
+ result = self.recover(action, result, ctx)
168
+ if result.success and action.type == "click":
169
+ self._pending_verify = self._capture_expected(obs, action)
170
+ self._save_artifact(obs, action, result)
171
+ self.events.publish(ActionFinished(result))
172
+ if self.history is not None:
173
+ self.history.record(action.id, action.type, result.success, result.message)
174
+ self.safety.record_step()
175
+ steps += 1
176
+ if action.type not in ("success", "fail"):
177
+ sig = f"{action.type}({sorted(action.params.items())})"
178
+ if self._recent_sigs and self._recent_sigs[-1] == sig:
179
+ repeat_count += 1
180
+ else:
181
+ repeat_count = 1
182
+ self._recent_sigs.append(sig)
183
+ if repeat_count >= 6:
184
+ finished_status = "FAIL"
185
+ finished_summary = f"stuck: repeated {sig} {repeat_count} times with no effect"
186
+ break
187
+ if action.type == "success":
188
+ blocker = self._unconfirmed_edit()
189
+ if blocker:
190
+ # hard guard: an element_id-less type (rename box /
191
+ # filename field) was not confirmed with Enter, so a
192
+ # success() would claim an edit that never applied.
193
+ hints.append(blocker)
194
+ self._save_artifact(obs, action, result)
195
+ self.events.publish(ActionFinished(ActionResult(
196
+ action.id, False, blocker, retryable=True)))
197
+ if self.history is not None:
198
+ self.history.record(action.id, action.type, False, blocker)
199
+ self.safety.record_step()
200
+ steps += 1
201
+ continue
202
+ finished_status = "SUCCESS"
203
+ finished_summary = str(action.params.get("result", ""))
204
+ break
205
+ if action.type == "fail":
206
+ finished_status = "FAIL"
207
+ finished_summary = str(action.params.get("reason", ""))
208
+ break
209
+ # Only ONE action per observation: the screen changes after
210
+ # every action, so subsequent actions in the same plan were
211
+ # decided against a stale scene. Re-observe + replan first.
212
+ # (Terminal success/fail already broke above.)
213
+ if i + 1 < len(plan.actions):
214
+ break
215
+ if finished_status in ("SUCCESS", "FAIL"):
216
+ break
217
+ prev = obs
218
+ except Exception as e:
219
+ finished_status = "FAIL"
220
+ finished_summary = f"loop error: {e}"
221
+ finally:
222
+ # capture status BEFORE stopping so stop() doesn't flip it to ABORTED
223
+ terminal = self.safety.status()
224
+ self.safety.stop()
225
+
226
+ status = finished_status or terminal
227
+ if status == "RUNNING":
228
+ status = "FAIL"
229
+ self._prune_artifacts()
230
+ result = TaskResult(
231
+ status=status,
232
+ summary=finished_summary,
233
+ task_id=self._task_id,
234
+ steps=steps,
235
+ duration=time.time() - start,
236
+ artifacts=self._artifact_paths,
237
+ )
238
+ self.events.publish(TaskFinished(result))
239
+ return result
240
+
241
+ def _capture_expected(self, obs, action):
242
+ """Record the clicked node's expected screen change for verification."""
243
+ scene = getattr(obs, "scene", None)
244
+ if scene is None:
245
+ return None
246
+ node_id = action.params.get("element_id")
247
+ if node_id is None:
248
+ return None
249
+ aff = scene.affordance_for(int(node_id), "click")
250
+ if aff is None or not aff.expected:
251
+ return None
252
+ return (int(node_id), dict(aff.expected), scene)
253
+
254
+ def _verify_pending(self, obs):
255
+ """Check whether the previous click produced its expected change."""
256
+ node_id, expected, prev_scene = self._pending_verify
257
+ curr_scene = getattr(obs, "scene", None)
258
+ if curr_scene is None:
259
+ return None
260
+ ok, detail = self._verifier.verify(prev_scene, curr_scene, expected)
261
+ if ok:
262
+ return None
263
+ return (f"VERIFICATION: your last click on node {node_id} did not have the "
264
+ f"expected effect ({detail}). It likely missed or the target changed. "
265
+ f"Do NOT repeat it blindly -- re-inspect and pick a fresh target.")
266
+
267
+ def _unconfirmed_edit(self):
268
+ """A rename/filename type (type WITHOUT element_id) that has NOT been
269
+ confirmed with Enter. success() must be blocked then, or the agent
270
+ claims an edit that never applied (folder/file name unchanged)."""
271
+ sigs = list(getattr(self, "_recent_sigs", None) or [])
272
+ typed = [s for s in sigs if s.startswith("type(") and "element_id" not in s]
273
+ if not typed:
274
+ return None
275
+ if any(self._is_confirming_action(s) for s in sigs):
276
+ return None
277
+ return ("BLOCKED: you typed a name but never pressed `enter` to apply it "
278
+ "-- the edit is NOT saved. Call `key(keys=\"enter\")` to confirm, "
279
+ "then call `success`.")
280
+
281
+ def _rename_hint(self):
282
+ """If the agent pressed ctrl+shift+n (created a new folder) but has not
283
+ typed a name for it, the folder stays '新建文件夹' and the task is not
284
+ done. Tell it to type the name (type WITHOUT element_id -- the rename
285
+ box has focus)."""
286
+ sigs = list(getattr(self, "_recent_sigs", None) or [])
287
+ if not sigs:
288
+ return None
289
+ created = any("ctrl+shift+n" in s for s in sigs[-4:])
290
+ if not created:
291
+ return None
292
+ if any(s.startswith("type(") for s in sigs[-4:]):
293
+ return None
294
+ return ("You created a new folder (ctrl+shift+n) but have NOT typed its "
295
+ "name. Its name is selected and the rename box has focus -- call "
296
+ "`type` WITHOUT element_id (e.g. `type(text=\"smoke_demo_folder\")`) "
297
+ "to name it, then press `enter`.")
298
+
299
+ def _confirm_hint(self):
300
+ """If the agent typed WITHOUT an element_id (a focused rename box or
301
+ filename field -- the explorer/filename case) and has not pressed Enter
302
+ to confirm, the edit is still pending. Tell it to press Enter instead
303
+ of re-typing or moving on. The classic explorer failure: type the
304
+ folder name, never press Enter, call success.
305
+
306
+ Types WITH an element_id (e.g. notepad body) do not need Enter, so we
307
+ only react to element_id-less types to avoid false positives.
308
+ """
309
+ sigs = list(getattr(self, "_recent_sigs", None) or [])
310
+ if not sigs:
311
+ return None
312
+ typed_without_target = [s for s in sigs
313
+ if s.startswith("type(") and "element_id" not in s]
314
+ if not typed_without_target:
315
+ return None
316
+ # ...and no Enter/Save confirmation after the latest such type
317
+ if any(self._is_confirming_action(s) for s in sigs[-3:]):
318
+ return None
319
+ return ("You typed into a rename/filename box (type without element_id) "
320
+ "but have NOT pressed `enter` to confirm it -- the change is NOT "
321
+ "applied until `enter`. Call `key(keys=\"enter\")` now, then the "
322
+ "task is done.")
323
+
324
+ def _completion_hint(self, no_change):
325
+ """Hint the agent to finish when it has done the closing steps but
326
+ keeps verifying instead of calling success (task completed but
327
+ status=TIMEOUT).
328
+
329
+ Requires a *confirming* action (Enter / Save) reasonably recent AND the
330
+ screen to have settled (no_change >= 1). Not just any type: typing
331
+ without Enter leaves the edit unconfirmed, so that alone is not
332
+ completion (see _confirm_hint). A confirming action later in the
333
+ trailing history means a save/rename was applied.
334
+ """
335
+ if no_change < 1:
336
+ return None
337
+ sigs = list(getattr(self, "_recent_sigs", None) or [])
338
+ if len(sigs) < 2:
339
+ return None
340
+ confirmed = [s for s in sigs if self._is_confirming_action(s)]
341
+ if not confirmed:
342
+ return None
343
+ # The confirming action should not be buried too far behind unrelated work.
344
+ last_index = max(i for i, s in enumerate(sigs) if self._is_confirming_action(s))
345
+ if len(sigs) - 1 - last_index > 2:
346
+ return None
347
+ return ("The screen has settled and you already pressed Enter / Save to "
348
+ "apply the change. If the task's goal is met, STOP and call "
349
+ "`success` with a summary now -- do not keep acting.")
350
+
351
+ @staticmethod
352
+ def _is_confirming_action(sig):
353
+ """Actions that CONFIRM a save/rename: Enter, Save click, Ctrl+S."""
354
+ if sig.startswith("key("):
355
+ return any(_keys_eq(sig, k) for k in ("enter", "ctrl+s"))
356
+ if sig.startswith("click("):
357
+ return "Save" in sig or "保存" in sig
358
+ return False