From 44a9dafcd81bfd5d55e45441495a0892a9a994b9 Mon Sep 17 00:00:00 2001 From: butubb <1422726308@qq.com> Date: Fri, 4 Sep 2026 14:29:17 +0800 Subject: [PATCH] =?UTF-8?q?fix:=20AI=20=E7=82=B9=E5=87=BB=E7=B2=BE?= =?UTF-8?q?=E5=BA=A6=E4=BC=98=E5=8C=96=E2=80=94=E2=80=94=E2=91=A0=E5=9D=90?= =?UTF-8?q?=E6=A0=87=E5=90=B8=E9=99=84=EF=BC=9A/api/screen/tap=20=E5=8A=A0?= =?UTF-8?q?=20snap=3D1=EF=BC=8CMCP=20de=5Ftap=20=E5=9B=BA=E5=AE=9A?= =?UTF-8?q?=E5=BC=80=E5=90=AF=EF=BC=88dump=20UI=20=E6=A0=91=E6=89=BE?= =?UTF-8?q?=E5=8C=85=E5=90=AB=E7=82=B9=E5=87=BB=E7=82=B9=E7=9A=84=E6=9C=80?= =?UTF-8?q?=E5=B0=8F=E5=8F=AF=E7=82=B9=E5=87=BB=E5=85=83=E7=B4=A0=E7=82=B9?= =?UTF-8?q?=E4=B8=AD=E5=BF=83=EF=BC=8C=E6=A8=A1=E5=9E=8B=E5=9D=90=E6=A0=87?= =?UTF-8?q?=E5=81=8F=2020-50px=20=E4=B9=9F=E7=82=B9=E5=BE=97=E5=87=86?= =?UTF-8?q?=EF=BC=8C=E5=A4=A7=E5=B1=8F=E8=A7=A6=E6=8E=A7=E4=B8=8D=E5=B8=A6?= =?UTF-8?q?=20snap=20=E8=A1=8C=E4=B8=BA=E4=B8=8D=E5=8F=98=EF=BC=89?= =?UTF-8?q?=EF=BC=9B=E2=91=A1=E6=96=B0=20de=5Ftap=5Ftext=20=E8=AF=AD?= =?UTF-8?q?=E4=B9=89=E7=82=B9=E5=87=BB=EF=BC=9A=E6=8C=89=E5=B1=8F=E5=B9=95?= =?UTF-8?q?=E5=8F=AF=E8=A7=81=E6=96=87=E5=AD=97=E4=B8=80=E6=AC=A1=E5=AE=8C?= =?UTF-8?q?=E6=88=90=E6=89=BE+=E7=82=B9=EF=BC=88UI=20=E6=A0=91=20textConta?= =?UTF-8?q?ins/descriptionContains=20=E2=86=92=20OCR=20=E4=B8=AD=E5=BF=83?= =?UTF-8?q?=E5=85=9C=E5=BA=95=EF=BC=8CWebView/=E5=9B=BE=E7=89=87=E6=96=87?= =?UTF-8?q?=E5=AD=97=E4=B9=9F=E8=83=BD=E7=82=B9=EF=BC=89=EF=BC=9B=E2=91=A2?= =?UTF-8?q?de=5Ftap=5Felement=20=E6=94=AF=E6=8C=81=20text=5Fcontains/desc?= =?UTF-8?q?=5Fcontains=20=E6=A8=A1=E7=B3=8A=E5=8C=B9=E9=85=8D=EF=BC=9B?= =?UTF-8?q?=E2=91=A3de=5Fui=5Ftree=20=E5=8F=AF=E7=82=B9=E5=87=BB=E5=85=83?= =?UTF-8?q?=E7=B4=A0=E4=BC=98=E5=85=88=20+=20limit=20=E5=8F=82=E6=95=B0?= =?UTF-8?q?=E9=98=B2=20token=20=E8=86=A8=E8=83=80=EF=BC=9B=E2=91=A4agent?= =?UTF-8?q?=20=E6=8F=90=E7=A4=BA=E8=AF=8D=E9=87=8D=E5=86=99=EF=BC=9A?= =?UTF-8?q?=E6=96=87=E5=AD=97=E8=AF=AD=E4=B9=89=E7=82=B9=E5=87=BB=E4=BC=98?= =?UTF-8?q?=E5=85=88=E3=80=81=E5=9D=90=E6=A0=87=E4=BB=85=E7=BA=AF=E5=9B=BE?= =?UTF-8?q?=E5=BD=A2=E5=85=9C=E5=BA=95=E4=B8=94=E8=87=AA=E5=8A=A8=E5=90=B8?= =?UTF-8?q?=E9=99=84=E3=80=81=E7=82=B9=E5=87=BB=E5=90=8E=E6=97=A0=E5=8F=98?= =?UTF-8?q?=E5=8C=96=E7=A6=81=E6=AD=A2=E9=87=8D=E5=A4=8D=E5=90=8C=E5=9D=90?= =?UTF-8?q?=E6=A0=87?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- mcp_agent/agent.py | 24 +++++--- mcp_server/mcp_server.py | 80 ++++++++++++++++++++++----- mcp_server/platform_client.py | 47 ++++++++-------- web/monitor.py | 101 +++++++++++++++++++++++++++++++++- 4 files changed, 203 insertions(+), 49 deletions(-) diff --git a/mcp_agent/agent.py b/mcp_agent/agent.py index ba80ac1..82f8368 100644 --- a/mcp_agent/agent.py +++ b/mcp_agent/agent.py @@ -27,13 +27,23 @@ SYSTEM_PROMPT = """你是手机自动化控制助手。你通过工具实时操 工作规范: 1. 先 de_list_devices 确定目标设备(在线才可操作) 2. 观察屏幕:先 de_screenshot 获取截图(图像会随后给你),基于截图理解当前界面 -3. 操作:de_tap/de_swipe 的坐标必须与最近一次 de_screenshot 图像一致(直接看图给坐标,服务器自动换算) -4. 元素操作优先:能用 de_ui_tree/de_tap_element(text/id/desc 定位)就不用裸坐标 -5. 每次关键操作后再次 de_screenshot 验证结果,直到完成用户目标 -6. 完成或失败时用中文总结:做了什么、当前状态、需要用户注意的事项 -7. 设备不可用/操作失败时如实报告错误,不要臆测成功 -8. 效率:界面未变化时不要重复截图/点击同一位置;每步都要推进目标; - 若连续 6 步无进展(截图内容未变/操作无效),停止并总结原因,不要空转 +3. 点击定位分优先级(不要自己推算像素坐标,那是精度最差的方式): + a) 目标有可见文字(按钮/菜单/列表标题/标签/输入框提示)→ de_tap_text 直接给文字, + 一次完成「找到并点击」,原生控件与 WebView/图片渲染文字都支持 + b) 文字有歧义或 de_tap_text 未命中 → de_ui_tree(limit=80) 看可点元素后 + 用 de_tap_element(text/text_contains 匹配) + c) 只有纯图形目标(视频画面/无文字图标且树里没有)才用 de_tap 给坐标—— + 坐标只需大致对准目标中心,服务端会自动吸附到该处可点击元素中心,无需精算 +4. de_tap 点击后若返回 snapped=true 表示已吸附命中元素(可核对 label); + 截图判断界面变化=点击成功,无变化=未命中 +5. 输入文字:先 de_tap_text 或 de_tap 点中输入框,再 de_type_text 输入 +6. 每次关键操作后再次 de_screenshot 验证结果,直到完成用户目标 +7. 若点击后截图无任何变化:不要重复点同一坐标,换 de_tap_text/de_tap_element + 重新定位,或先 de_ui_tree 确认元素文案再试 +8. 完成或失败时用中文总结:做了什么、当前状态、需要用户注意的事项 +9. 设备不可用/操作失败时如实报告错误,不要臆测成功 +10. 效率:界面未变化时不要重复截图/点击同一位置;每步都要推进目标; + 若连续 6 步无进展(截图内容未变/操作无效),停止并总结原因,不要空转 可用工具清单将由系统提供。""" diff --git a/mcp_server/mcp_server.py b/mcp_server/mcp_server.py index b22c152..2abbbda 100644 --- a/mcp_server/mcp_server.py +++ b/mcp_server/mcp_server.py @@ -129,10 +129,11 @@ def de_screenshot(serial: str) -> dict: @mcp.tool() def de_tap(serial: str, x: int, y: int) -> dict: - """点击设备屏幕指定坐标。 + """点击设备屏幕指定坐标(坐标空间 = de_screenshot 的图像坐标)。 - 坐标空间 = de_screenshot 返回的图像坐标(display 空间)——先截图拿到 - native_size 后再点击,server 自动换算为设备原生坐标。 + 自动吸附:若该点落在某个可点击元素内,实际点击会改为该元素的中心—— + 坐标只需大致对准目标即可(模型视觉定位常有偏差,吸附保证点准); + 点在空白处则按原坐标点击。返回中的 snapped/label 可核对吸附结果。 """ try: _check_write() @@ -140,11 +141,16 @@ def de_tap(serial: str, x: int, y: int) -> dict: if x < 0 or y < 0: raise PlatformError("invalid_param", "坐标不能为负") nx, ny = _to_native(serial, x, y) - platform().tap(serial, nx, ny) + res = platform().tap(serial, nx, ny, snap=True) except PlatformError as e: return _err(e) - audit.audit("de_tap", serial, f"({x},{y})->native({nx},{ny})", "ok") - return _ok({"action": "tap", "serial": serial, "x": x, "y": y}) + audit.audit("de_tap", serial, + f"({x},{y})->native({nx},{ny})" + + (f" 吸附[{res.get('label')}]" if res.get("snapped") else ""), + "ok") + return _ok({"action": "tap", "serial": serial, "x": x, "y": y, + "snapped": bool(res.get("snapped")), + "label": res.get("label") or ""}) @mcp.tool() @@ -168,17 +174,21 @@ _KEYS = ("back", "home", "recent", "menu", "power", "volume_up", @mcp.tool() -def de_ui_tree(serial: str) -> dict: +def de_ui_tree(serial: str, limit: int = 150) -> dict: """获取当前界面元素树(文本 JSON):每元素含 text/resource_id/description/class/bounds。 - 优先用它定位元素(元素驱动操作),比纯坐标点击更可靠。 + 可点击元素排在前面(可点性优先)。多数场景不需要读整棵树——直接给 + de_tap_text 一个屏幕上可见的文字即可自动定位点击;本工具用于确认界面 + 上有什么、元素文案是否与预想一致。limit 控制返回条数(默认 150,防 token 膨胀)。 """ try: serial = _check_serial(serial) + if limit < 1 or limit > 300: + raise PlatformError("invalid_param", "limit 需在 1-300 之间") els = platform().ui_elements(serial) except PlatformError as e: return _err(e) - # 精简输出:去掉 suggested/深度噪音,保留可定位属性 + # 精简输出:去掉 suggested/深度噪音,保留可定位属性;可点击优先、有文案优先 slim = [] for e in els: slim.append({ @@ -186,30 +196,39 @@ def de_ui_tree(serial: str) -> dict: "id": e.get("resource_id", "")[:80], "desc": e.get("description", "")[:50], "class": e.get("class", "").split(".")[-1], + "clickable": e.get("clickable", "") == "true", "bounds": e.get("bounds", ""), }) + slim.sort(key=lambda x: (not x["clickable"], not (x["text"] or x["desc"]))) audit.audit("de_ui_tree", serial, "", f"{len(slim)} 元素") - return _ok({"count": len(slim), "elements": slim[:300]}) + return _ok({"count": len(slim), "elements": slim[:limit]}) @mcp.tool() def de_tap_element(serial: str, by: str, value: str, index: int = 1) -> dict: - """按元素点击(不需要坐标):by=text|id|desc,value 为匹配文本/资源 id/描述。 + """按元素点击(不需要坐标):by=text|id|desc|text_contains|desc_contains。 + text/id/desc 为精确匹配;text_contains/desc_contains 为子串模糊匹配 + (只记得部分文字时用,如 by=text_contains value=搜索)。 元素驱动操作比坐标可靠(界面变化自适应);元素不存在时返回错误, - 可改用 de_ui_tree 查元素或 de_tap 坐标兜底。index 用于多命中取第几个(默认 1)。 + 可改用 de_ui_tree 查元素 / de_tap_text 按屏幕文字点 / de_tap 坐标兜底。 + index 用于多命中取第几个(默认 1)。 """ try: _check_write() serial = _check_serial(serial) - if by not in ("text", "id", "desc"): - raise PlatformError("invalid_param", "by 可选 text/id/desc") + if by not in ("text", "id", "desc", "text_contains", "desc_contains"): + raise PlatformError("invalid_param", + "by 可选 text/id/desc/text_contains/desc_contains") if not value or index < 1: raise PlatformError("invalid_param", "value 不能为空且 index>=1") import uiautomator2 as u2 d = u2.connect(serial) kw = {"text": value} if by == "text" else ( - {"resourceId": value} if by == "id" else {"description": value}) + {"resourceId": value} if by == "id" else ( + {"description": value} if by == "desc" else ( + {"textContains": value} if by == "text_contains" + else {"descriptionContains": value}))) if index > 1: kw["instance"] = index - 1 el = d(**kw) @@ -397,6 +416,37 @@ def de_ocr(serial: str) -> dict: return _ok({"count": len(slim), "texts": slim[:100]}) +@mcp.tool() +def de_tap_text(serial: str, text: str) -> dict: + """点击屏幕上显示该文字的位置(语义点击:一次调用完成「找到并点击」,无需坐标)。 + + 想点带文字的按钮/列表项/标签/链接时用它:text 只需是屏幕上可见文字的 + 一部分(子串匹配,如「搜索」「立即购买」)。原生控件直接命中; + WebView/图片/画布里渲染的文字自动走 OCR 兜底。多命中点第一处(想点 + 更靠下的请把文字换独特些)。屏幕确实没有该文字时返回错误提示, + 请截图确认后换关键词。比 de_tap 坐标点击可靠,涉及文字目标时优先使用。 + """ + try: + _check_write() + serial = _check_serial(serial) + if not text or len(text) > 100: + raise PlatformError("invalid_param", "text 不能为空且 ≤100 字符") + res = platform().tap_text(serial, text) + except PlatformError as e: + return _err(e) + if not res.get("found"): + err = PlatformError("text_not_found", + f"屏幕上未找到文字「{text}」——先 de_screenshot 看当前界面," + f"换用屏幕上实际存在的文字;若文字在需滑动后才可见请先滑动") + audit.audit("de_tap_text", serial, f"「{text[:30]}」", "未找到") + return _err(err) + audit.audit("de_tap_text", serial, + f"「{text[:30]}」via {res.get('method')} @({res.get('x')},{res.get('y')})", "ok") + return _ok({"action": "tap_text", "serial": serial, "text": text, + "method": res.get("method"), "matched": res.get("matched") or text, + "x": res.get("x"), "y": res.get("y")}) + + @mcp.tool() def de_list_apps(serial: str, keyword: str = "") -> dict: """列出设备第三方已装应用包名(可关键词过滤,如 keyword='douyin' 找抖音)。""" diff --git a/mcp_server/platform_client.py b/mcp_server/platform_client.py index c14bdea..1fea8f9 100644 --- a/mcp_server/platform_client.py +++ b/mcp_server/platform_client.py @@ -93,12 +93,32 @@ class PlatformClient: raise PlatformError("device_offline", str(j.get("error", "取分辨率失败"))[:120]) return int(j["width"]), int(j["height"]) - def tap(self, serial, x, y): - """点击(POST /api/screen/tap)。""" + def tap(self, serial, x, y, snap=False): + """点击(POST /api/screen/tap)。 + + snap=True:点落在可点击元素内则吸附到元素中心(AI 粗略坐标也能点准)。 + 返回平台 JSON(含 snapped/x/y/label)。 + """ r = self._post("/api/screen/tap", json={"serial": serial, - "x": int(x), "y": int(y)}) + "x": int(x), "y": int(y), + "snap": 1 if snap else 0}) return self._check_op(r, "tap") + def tap_text(self, serial, text): + """按屏幕文字点击(平台解析:UI 树子串匹配 → OCR 兜底)。 + + 返回 {ok, found, method, matched, x, y}——found=false 是业务结果 + (屏幕无该文字),非设备错误;设备离线/不可达仍抛 PlatformError。 + """ + r = self._post("/api/screen/tap_text", + json={"serial": serial, "text": str(text)}) + if r.status_code == 503: + raise PlatformError("device_offline", r.text[:120]) + if r.status_code != 200: + raise PlatformError("platform_unavailable", + f"tap_text HTTP {r.status_code}: {r.text[:120]}") + return r.json() or {} + def swipe(self, serial, x1, y1, x2, y2, duration=0.2): """滑动(POST /api/screen/swipe)。""" r = self._post("/api/screen/swipe", json={ @@ -147,27 +167,6 @@ class PlatformClient: "schedule": (t.get("schedule") or {}).get("mode", "")}) return tasks - def sleep(self, serial): - """熄屏(POST /api/device/screen_all mode=off)。""" - r = self._post("/api/device/screen_all", - json={"mode": "off", "serials": [serial]}) - return self._check_op(r, "sleep") - - def list_tasks(self): - """任务计划列表(GET /api/jobs)。""" - r = self._get("/api/jobs") - if r.status_code != 200: - raise PlatformError("platform_unavailable", - f"/api/jobs HTTP {r.status_code}") - j = r.json() or {} - tasks = [] - for t in j.get("jobs") or []: - tasks.append({"id": t.get("id"), "name": t.get("name"), - "task_type": t.get("task_type"), - "enabled": t.get("enabled"), - "schedule": (t.get("schedule") or {}).get("mode", "")}) - return tasks - def press_key(self, serial, key): """按键(POST /api/screen/key)。""" r = self._post("/api/screen/key", diff --git a/web/monitor.py b/web/monitor.py index dcfca73..4731b6d 100644 --- a/web/monitor.py +++ b/web/monitor.py @@ -1,6 +1,7 @@ """监控域 API:状态/运行控制/设备操作/远程看屏。""" import time import threading +import re import subprocess import shlex from concurrent.futures import ThreadPoolExecutor, as_completed @@ -476,21 +477,115 @@ def api_screen_thumb(): return jsonify({"ok": False, "error": str(e)}), 503 +def _snap_to_clickable(d, x, y): + """坐标吸附:找包含 (x,y) 的最小可点击元素,返回其 bounds 中心。 + + AI/触控给的坐标常偏离目标 20-50px(模型视觉定位误差)。dump 当前 UI 树后 + 遍历 clickable 节点:点击点落在哪个可点元素内就点它的中心——偏了也点得准; + 取包含元素中面积最小者(最具体的那个)。点空白处(收键盘等)无包含元素则 + 原坐标返回。dump 失败也不阻塞,直接原坐标。返回 (cx, cy, snapped, label)。 + """ + try: + import xml.etree.ElementTree as ET + xml_str = d.dump_hierarchy() + best = None # (area, cx, cy, label) + for node in ET.fromstring(xml_str).iter("node"): + if node.get("clickable") != "true": + continue + if node.get("enabled") == "false": + continue + m = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", node.get("bounds", "")) + if not m: + continue + x1, y1, x2, y2 = map(int, m.groups()) + if x1 <= x <= x2 and y1 <= y <= y2: + area = (x2 - x1) * (y2 - y1) + if best is None or area < best[0]: + best = (area, (x1 + x2) // 2, (y1 + y2) // 2, + (node.get("text") or node.get("content-desc") or "")[:40]) + if best: + return best[1], best[2], True, best[3] + except Exception: + pass + return x, y, False, "" + + @bp.route("/api/screen/tap", methods=["POST"]) @perm_required(PERM_DEVICES) def api_screen_tap(): - """点击:{serial, x, y}(设备原生分辨率坐标)。""" + """点击:{serial, x, y, snap?}(设备原生分辨率坐标)。 + + snap=1 时先吸附:点落在可点击元素内则改点元素中心(MCP/AI 场景用,粗略 + 坐标也能点准);大屏精确触控不带 snap 保持原行为。 + 返回 {ok, snapped, x, y, label}。 + """ data = request.json or {} serial, x, y = data.get("serial", ""), data.get("x"), data.get("y") + snap = int(data.get("snap", 0) or 0) == 1 if not serial or x is None or y is None: return jsonify({"ok": False, "error": "缺少 serial/x/y"}), 400 try: - _screen_get_device(serial).click(int(x), int(y)) - return jsonify({"ok": True}) + d = _screen_get_device(serial) + sx, sy = int(x), int(y) + label = "" + snapped = False + if snap: + sx, sy, snapped, label = _snap_to_clickable(d, sx, sy) + d.click(sx, sy) + return jsonify({"ok": True, "snapped": snapped, "x": sx, "y": sy, + "label": label}) except Exception as e: _screen_invalidate(serial) return jsonify({"ok": False, "error": f"点击失败: {e}"}), 503 + +@bp.route("/api/screen/tap_text", methods=["POST"]) +@perm_required(PERM_DEVICES) +def api_screen_tap_text(): + """按屏幕文字点击:{serial, text}——找到显示该文字的位置并点中心(子串匹配)。 + + 两步:① UI 树 textContains/descriptionContains 命中 → 点元素中心(原生控件); + ② 未命中 → 截图 OCR 找文字中心(WebView/图片/画布渲染的文字)。 + 返回 {ok, found, method: ui|ocr, matched, x, y};found=false 表示屏幕确实 + 没有该文字(业务结果非设备错误);OCR 不可用且 UI 未命中时也返回 found=false。 + """ + data = request.json or {} + serial = (data.get("serial") or "").strip() + text = (data.get("text") or "").strip() + if not serial or not text: + return jsonify({"ok": False, "error": "缺少 serial/text"}), 400 + if len(text) > 100: + return jsonify({"ok": False, "error": "文字过长(≤100 字符)"}), 400 + try: + d = _screen_get_device(serial) + # ① UI 树:text / content-desc 子串匹配(原生控件最快最准) + for kw in ({"textContains": text}, {"descriptionContains": text}): + try: + el = d(**kw) + if el.exists(timeout=1.5): + b = el.bounds + el.click() + return jsonify({"ok": True, "found": True, "method": "ui", + "matched": text, + "x": (b[0] + b[2]) // 2, "y": (b[1] + b[3]) // 2}) + except Exception: + continue + # ② OCR 兜底:截图找文字中心(UI 树没有的渲染文字) + img = d.screenshot() + if img is not None: + from core.ocr import find_on_screen + found, center, matched = find_on_screen(img, text) + if found: + d.click(*center) + return jsonify({"ok": True, "found": True, "method": "ocr", + "matched": matched, + "x": center[0], "y": center[1]}) + return jsonify({"ok": True, "found": False, "method": "", + "matched": "", "x": 0, "y": 0}) + except Exception as e: + _screen_invalidate(serial) + return jsonify({"ok": False, "error": f"文字点击失败: {e}"}), 503 + @bp.route("/api/screen/swipe", methods=["POST"]) @perm_required(PERM_DEVICES) def api_screen_swipe():