fix: AI 点击精度优化——①坐标吸附:/api/screen/tap 加 snap=1,MCP de_tap 固定开启(dump UI 树找包含点击点的最小可点击元素点中心,模型坐标偏 20-50px 也点得准,大屏触控不带 snap 行为不变);②新 de_tap_text 语义点击:按屏幕可见文字一次完成找+点(UI 树 textContains/descriptionContains → OCR 中心兜底,WebView/图片文字也能点);③de_tap_element 支持 text_contains/desc_contains 模糊匹配;④de_ui_tree 可点击元素优先 + limit 参数防 token 膨胀;⑤agent 提示词重写:文字语义点击优先、坐标仅纯图形兜底且自动吸附、点击后无变化禁止重复同坐标
This commit is contained in:
+17
-7
@@ -27,13 +27,23 @@ SYSTEM_PROMPT = """你是手机自动化控制助手。你通过工具实时操
|
||||
工作规范:
|
||||
1. 先 de_list_devices 确定目标设备(在线才可操作)
|
||||
2. 观察屏幕:先 de_screenshot 获取截图(图像会随后给你),基于截图理解当前界面
|
||||
3. 操作:de_tap/de_swipe 的坐标必须与最近一次 de_screenshot 图像一致(直接看图给坐标,服务器自动换算)
|
||||
4. 元素操作优先:能用 de_ui_tree/de_tap_element(text/id/desc 定位)就不用裸坐标
|
||||
5. 每次关键操作后再次 de_screenshot 验证结果,直到完成用户目标
|
||||
6. 完成或失败时用中文总结:做了什么、当前状态、需要用户注意的事项
|
||||
7. 设备不可用/操作失败时如实报告错误,不要臆测成功
|
||||
8. 效率:界面未变化时不要重复截图/点击同一位置;每步都要推进目标;
|
||||
若连续 6 步无进展(截图内容未变/操作无效),停止并总结原因,不要空转
|
||||
3. 点击定位分优先级(不要自己推算像素坐标,那是精度最差的方式):
|
||||
a) 目标有可见文字(按钮/菜单/列表标题/标签/输入框提示)→ de_tap_text 直接给文字,
|
||||
一次完成「找到并点击」,原生控件与 WebView/图片渲染文字都支持
|
||||
b) 文字有歧义或 de_tap_text 未命中 → de_ui_tree(limit=80) 看可点元素后
|
||||
用 de_tap_element(text/text_contains 匹配)
|
||||
c) 只有纯图形目标(视频画面/无文字图标且树里没有)才用 de_tap 给坐标——
|
||||
坐标只需大致对准目标中心,服务端会自动吸附到该处可点击元素中心,无需精算
|
||||
4. de_tap 点击后若返回 snapped=true 表示已吸附命中元素(可核对 label);
|
||||
截图判断界面变化=点击成功,无变化=未命中
|
||||
5. 输入文字:先 de_tap_text 或 de_tap 点中输入框,再 de_type_text 输入
|
||||
6. 每次关键操作后再次 de_screenshot 验证结果,直到完成用户目标
|
||||
7. 若点击后截图无任何变化:不要重复点同一坐标,换 de_tap_text/de_tap_element
|
||||
重新定位,或先 de_ui_tree 确认元素文案再试
|
||||
8. 完成或失败时用中文总结:做了什么、当前状态、需要用户注意的事项
|
||||
9. 设备不可用/操作失败时如实报告错误,不要臆测成功
|
||||
10. 效率:界面未变化时不要重复截图/点击同一位置;每步都要推进目标;
|
||||
若连续 6 步无进展(截图内容未变/操作无效),停止并总结原因,不要空转
|
||||
|
||||
可用工具清单将由系统提供。"""
|
||||
|
||||
|
||||
+65
-15
@@ -129,10 +129,11 @@ def de_screenshot(serial: str) -> dict:
|
||||
|
||||
@mcp.tool()
|
||||
def de_tap(serial: str, x: int, y: int) -> dict:
|
||||
"""点击设备屏幕指定坐标。
|
||||
"""点击设备屏幕指定坐标(坐标空间 = de_screenshot 的图像坐标)。
|
||||
|
||||
坐标空间 = de_screenshot 返回的图像坐标(display 空间)——先截图拿到
|
||||
native_size 后再点击,server 自动换算为设备原生坐标。
|
||||
自动吸附:若该点落在某个可点击元素内,实际点击会改为该元素的中心——
|
||||
坐标只需大致对准目标即可(模型视觉定位常有偏差,吸附保证点准);
|
||||
点在空白处则按原坐标点击。返回中的 snapped/label 可核对吸附结果。
|
||||
"""
|
||||
try:
|
||||
_check_write()
|
||||
@@ -140,11 +141,16 @@ def de_tap(serial: str, x: int, y: int) -> dict:
|
||||
if x < 0 or y < 0:
|
||||
raise PlatformError("invalid_param", "坐标不能为负")
|
||||
nx, ny = _to_native(serial, x, y)
|
||||
platform().tap(serial, nx, ny)
|
||||
res = platform().tap(serial, nx, ny, snap=True)
|
||||
except PlatformError as e:
|
||||
return _err(e)
|
||||
audit.audit("de_tap", serial, f"({x},{y})->native({nx},{ny})", "ok")
|
||||
return _ok({"action": "tap", "serial": serial, "x": x, "y": y})
|
||||
audit.audit("de_tap", serial,
|
||||
f"({x},{y})->native({nx},{ny})"
|
||||
+ (f" 吸附[{res.get('label')}]" if res.get("snapped") else ""),
|
||||
"ok")
|
||||
return _ok({"action": "tap", "serial": serial, "x": x, "y": y,
|
||||
"snapped": bool(res.get("snapped")),
|
||||
"label": res.get("label") or ""})
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
@@ -168,17 +174,21 @@ _KEYS = ("back", "home", "recent", "menu", "power", "volume_up",
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def de_ui_tree(serial: str) -> dict:
|
||||
def de_ui_tree(serial: str, limit: int = 150) -> dict:
|
||||
"""获取当前界面元素树(文本 JSON):每元素含 text/resource_id/description/class/bounds。
|
||||
|
||||
优先用它定位元素(元素驱动操作),比纯坐标点击更可靠。
|
||||
可点击元素排在前面(可点性优先)。多数场景不需要读整棵树——直接给
|
||||
de_tap_text 一个屏幕上可见的文字即可自动定位点击;本工具用于确认界面
|
||||
上有什么、元素文案是否与预想一致。limit 控制返回条数(默认 150,防 token 膨胀)。
|
||||
"""
|
||||
try:
|
||||
serial = _check_serial(serial)
|
||||
if limit < 1 or limit > 300:
|
||||
raise PlatformError("invalid_param", "limit 需在 1-300 之间")
|
||||
els = platform().ui_elements(serial)
|
||||
except PlatformError as e:
|
||||
return _err(e)
|
||||
# 精简输出:去掉 suggested/深度噪音,保留可定位属性
|
||||
# 精简输出:去掉 suggested/深度噪音,保留可定位属性;可点击优先、有文案优先
|
||||
slim = []
|
||||
for e in els:
|
||||
slim.append({
|
||||
@@ -186,30 +196,39 @@ def de_ui_tree(serial: str) -> dict:
|
||||
"id": e.get("resource_id", "")[:80],
|
||||
"desc": e.get("description", "")[:50],
|
||||
"class": e.get("class", "").split(".")[-1],
|
||||
"clickable": e.get("clickable", "") == "true",
|
||||
"bounds": e.get("bounds", ""),
|
||||
})
|
||||
slim.sort(key=lambda x: (not x["clickable"], not (x["text"] or x["desc"])))
|
||||
audit.audit("de_ui_tree", serial, "", f"{len(slim)} 元素")
|
||||
return _ok({"count": len(slim), "elements": slim[:300]})
|
||||
return _ok({"count": len(slim), "elements": slim[:limit]})
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def de_tap_element(serial: str, by: str, value: str, index: int = 1) -> dict:
|
||||
"""按元素点击(不需要坐标):by=text|id|desc,value 为匹配文本/资源 id/描述。
|
||||
"""按元素点击(不需要坐标):by=text|id|desc|text_contains|desc_contains。
|
||||
|
||||
text/id/desc 为精确匹配;text_contains/desc_contains 为子串模糊匹配
|
||||
(只记得部分文字时用,如 by=text_contains value=搜索)。
|
||||
元素驱动操作比坐标可靠(界面变化自适应);元素不存在时返回错误,
|
||||
可改用 de_ui_tree 查元素或 de_tap 坐标兜底。index 用于多命中取第几个(默认 1)。
|
||||
可改用 de_ui_tree 查元素 / de_tap_text 按屏幕文字点 / de_tap 坐标兜底。
|
||||
index 用于多命中取第几个(默认 1)。
|
||||
"""
|
||||
try:
|
||||
_check_write()
|
||||
serial = _check_serial(serial)
|
||||
if by not in ("text", "id", "desc"):
|
||||
raise PlatformError("invalid_param", "by 可选 text/id/desc")
|
||||
if by not in ("text", "id", "desc", "text_contains", "desc_contains"):
|
||||
raise PlatformError("invalid_param",
|
||||
"by 可选 text/id/desc/text_contains/desc_contains")
|
||||
if not value or index < 1:
|
||||
raise PlatformError("invalid_param", "value 不能为空且 index>=1")
|
||||
import uiautomator2 as u2
|
||||
d = u2.connect(serial)
|
||||
kw = {"text": value} if by == "text" else (
|
||||
{"resourceId": value} if by == "id" else {"description": value})
|
||||
{"resourceId": value} if by == "id" else (
|
||||
{"description": value} if by == "desc" else (
|
||||
{"textContains": value} if by == "text_contains"
|
||||
else {"descriptionContains": value})))
|
||||
if index > 1:
|
||||
kw["instance"] = index - 1
|
||||
el = d(**kw)
|
||||
@@ -397,6 +416,37 @@ def de_ocr(serial: str) -> dict:
|
||||
return _ok({"count": len(slim), "texts": slim[:100]})
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def de_tap_text(serial: str, text: str) -> dict:
|
||||
"""点击屏幕上显示该文字的位置(语义点击:一次调用完成「找到并点击」,无需坐标)。
|
||||
|
||||
想点带文字的按钮/列表项/标签/链接时用它:text 只需是屏幕上可见文字的
|
||||
一部分(子串匹配,如「搜索」「立即购买」)。原生控件直接命中;
|
||||
WebView/图片/画布里渲染的文字自动走 OCR 兜底。多命中点第一处(想点
|
||||
更靠下的请把文字换独特些)。屏幕确实没有该文字时返回错误提示,
|
||||
请截图确认后换关键词。比 de_tap 坐标点击可靠,涉及文字目标时优先使用。
|
||||
"""
|
||||
try:
|
||||
_check_write()
|
||||
serial = _check_serial(serial)
|
||||
if not text or len(text) > 100:
|
||||
raise PlatformError("invalid_param", "text 不能为空且 ≤100 字符")
|
||||
res = platform().tap_text(serial, text)
|
||||
except PlatformError as e:
|
||||
return _err(e)
|
||||
if not res.get("found"):
|
||||
err = PlatformError("text_not_found",
|
||||
f"屏幕上未找到文字「{text}」——先 de_screenshot 看当前界面,"
|
||||
f"换用屏幕上实际存在的文字;若文字在需滑动后才可见请先滑动")
|
||||
audit.audit("de_tap_text", serial, f"「{text[:30]}」", "未找到")
|
||||
return _err(err)
|
||||
audit.audit("de_tap_text", serial,
|
||||
f"「{text[:30]}」via {res.get('method')} @({res.get('x')},{res.get('y')})", "ok")
|
||||
return _ok({"action": "tap_text", "serial": serial, "text": text,
|
||||
"method": res.get("method"), "matched": res.get("matched") or text,
|
||||
"x": res.get("x"), "y": res.get("y")})
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def de_list_apps(serial: str, keyword: str = "") -> dict:
|
||||
"""列出设备第三方已装应用包名(可关键词过滤,如 keyword='douyin' 找抖音)。"""
|
||||
|
||||
@@ -93,12 +93,32 @@ class PlatformClient:
|
||||
raise PlatformError("device_offline", str(j.get("error", "取分辨率失败"))[:120])
|
||||
return int(j["width"]), int(j["height"])
|
||||
|
||||
def tap(self, serial, x, y):
|
||||
"""点击(POST /api/screen/tap)。"""
|
||||
def tap(self, serial, x, y, snap=False):
|
||||
"""点击(POST /api/screen/tap)。
|
||||
|
||||
snap=True:点落在可点击元素内则吸附到元素中心(AI 粗略坐标也能点准)。
|
||||
返回平台 JSON(含 snapped/x/y/label)。
|
||||
"""
|
||||
r = self._post("/api/screen/tap", json={"serial": serial,
|
||||
"x": int(x), "y": int(y)})
|
||||
"x": int(x), "y": int(y),
|
||||
"snap": 1 if snap else 0})
|
||||
return self._check_op(r, "tap")
|
||||
|
||||
def tap_text(self, serial, text):
|
||||
"""按屏幕文字点击(平台解析:UI 树子串匹配 → OCR 兜底)。
|
||||
|
||||
返回 {ok, found, method, matched, x, y}——found=false 是业务结果
|
||||
(屏幕无该文字),非设备错误;设备离线/不可达仍抛 PlatformError。
|
||||
"""
|
||||
r = self._post("/api/screen/tap_text",
|
||||
json={"serial": serial, "text": str(text)})
|
||||
if r.status_code == 503:
|
||||
raise PlatformError("device_offline", r.text[:120])
|
||||
if r.status_code != 200:
|
||||
raise PlatformError("platform_unavailable",
|
||||
f"tap_text HTTP {r.status_code}: {r.text[:120]}")
|
||||
return r.json() or {}
|
||||
|
||||
def swipe(self, serial, x1, y1, x2, y2, duration=0.2):
|
||||
"""滑动(POST /api/screen/swipe)。"""
|
||||
r = self._post("/api/screen/swipe", json={
|
||||
@@ -147,27 +167,6 @@ class PlatformClient:
|
||||
"schedule": (t.get("schedule") or {}).get("mode", "")})
|
||||
return tasks
|
||||
|
||||
def sleep(self, serial):
|
||||
"""熄屏(POST /api/device/screen_all mode=off)。"""
|
||||
r = self._post("/api/device/screen_all",
|
||||
json={"mode": "off", "serials": [serial]})
|
||||
return self._check_op(r, "sleep")
|
||||
|
||||
def list_tasks(self):
|
||||
"""任务计划列表(GET /api/jobs)。"""
|
||||
r = self._get("/api/jobs")
|
||||
if r.status_code != 200:
|
||||
raise PlatformError("platform_unavailable",
|
||||
f"/api/jobs HTTP {r.status_code}")
|
||||
j = r.json() or {}
|
||||
tasks = []
|
||||
for t in j.get("jobs") or []:
|
||||
tasks.append({"id": t.get("id"), "name": t.get("name"),
|
||||
"task_type": t.get("task_type"),
|
||||
"enabled": t.get("enabled"),
|
||||
"schedule": (t.get("schedule") or {}).get("mode", "")})
|
||||
return tasks
|
||||
|
||||
def press_key(self, serial, key):
|
||||
"""按键(POST /api/screen/key)。"""
|
||||
r = self._post("/api/screen/key",
|
||||
|
||||
+98
-3
@@ -1,6 +1,7 @@
|
||||
"""监控域 API:状态/运行控制/设备操作/远程看屏。"""
|
||||
import time
|
||||
import threading
|
||||
import re
|
||||
import subprocess
|
||||
import shlex
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
@@ -476,21 +477,115 @@ def api_screen_thumb():
|
||||
return jsonify({"ok": False, "error": str(e)}), 503
|
||||
|
||||
|
||||
def _snap_to_clickable(d, x, y):
|
||||
"""坐标吸附:找包含 (x,y) 的最小可点击元素,返回其 bounds 中心。
|
||||
|
||||
AI/触控给的坐标常偏离目标 20-50px(模型视觉定位误差)。dump 当前 UI 树后
|
||||
遍历 clickable 节点:点击点落在哪个可点元素内就点它的中心——偏了也点得准;
|
||||
取包含元素中面积最小者(最具体的那个)。点空白处(收键盘等)无包含元素则
|
||||
原坐标返回。dump 失败也不阻塞,直接原坐标。返回 (cx, cy, snapped, label)。
|
||||
"""
|
||||
try:
|
||||
import xml.etree.ElementTree as ET
|
||||
xml_str = d.dump_hierarchy()
|
||||
best = None # (area, cx, cy, label)
|
||||
for node in ET.fromstring(xml_str).iter("node"):
|
||||
if node.get("clickable") != "true":
|
||||
continue
|
||||
if node.get("enabled") == "false":
|
||||
continue
|
||||
m = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", node.get("bounds", ""))
|
||||
if not m:
|
||||
continue
|
||||
x1, y1, x2, y2 = map(int, m.groups())
|
||||
if x1 <= x <= x2 and y1 <= y <= y2:
|
||||
area = (x2 - x1) * (y2 - y1)
|
||||
if best is None or area < best[0]:
|
||||
best = (area, (x1 + x2) // 2, (y1 + y2) // 2,
|
||||
(node.get("text") or node.get("content-desc") or "")[:40])
|
||||
if best:
|
||||
return best[1], best[2], True, best[3]
|
||||
except Exception:
|
||||
pass
|
||||
return x, y, False, ""
|
||||
|
||||
|
||||
@bp.route("/api/screen/tap", methods=["POST"])
|
||||
@perm_required(PERM_DEVICES)
|
||||
def api_screen_tap():
|
||||
"""点击:{serial, x, y}(设备原生分辨率坐标)。"""
|
||||
"""点击:{serial, x, y, snap?}(设备原生分辨率坐标)。
|
||||
|
||||
snap=1 时先吸附:点落在可点击元素内则改点元素中心(MCP/AI 场景用,粗略
|
||||
坐标也能点准);大屏精确触控不带 snap 保持原行为。
|
||||
返回 {ok, snapped, x, y, label}。
|
||||
"""
|
||||
data = request.json or {}
|
||||
serial, x, y = data.get("serial", ""), data.get("x"), data.get("y")
|
||||
snap = int(data.get("snap", 0) or 0) == 1
|
||||
if not serial or x is None or y is None:
|
||||
return jsonify({"ok": False, "error": "缺少 serial/x/y"}), 400
|
||||
try:
|
||||
_screen_get_device(serial).click(int(x), int(y))
|
||||
return jsonify({"ok": True})
|
||||
d = _screen_get_device(serial)
|
||||
sx, sy = int(x), int(y)
|
||||
label = ""
|
||||
snapped = False
|
||||
if snap:
|
||||
sx, sy, snapped, label = _snap_to_clickable(d, sx, sy)
|
||||
d.click(sx, sy)
|
||||
return jsonify({"ok": True, "snapped": snapped, "x": sx, "y": sy,
|
||||
"label": label})
|
||||
except Exception as e:
|
||||
_screen_invalidate(serial)
|
||||
return jsonify({"ok": False, "error": f"点击失败: {e}"}), 503
|
||||
|
||||
|
||||
@bp.route("/api/screen/tap_text", methods=["POST"])
|
||||
@perm_required(PERM_DEVICES)
|
||||
def api_screen_tap_text():
|
||||
"""按屏幕文字点击:{serial, text}——找到显示该文字的位置并点中心(子串匹配)。
|
||||
|
||||
两步:① UI 树 textContains/descriptionContains 命中 → 点元素中心(原生控件);
|
||||
② 未命中 → 截图 OCR 找文字中心(WebView/图片/画布渲染的文字)。
|
||||
返回 {ok, found, method: ui|ocr, matched, x, y};found=false 表示屏幕确实
|
||||
没有该文字(业务结果非设备错误);OCR 不可用且 UI 未命中时也返回 found=false。
|
||||
"""
|
||||
data = request.json or {}
|
||||
serial = (data.get("serial") or "").strip()
|
||||
text = (data.get("text") or "").strip()
|
||||
if not serial or not text:
|
||||
return jsonify({"ok": False, "error": "缺少 serial/text"}), 400
|
||||
if len(text) > 100:
|
||||
return jsonify({"ok": False, "error": "文字过长(≤100 字符)"}), 400
|
||||
try:
|
||||
d = _screen_get_device(serial)
|
||||
# ① UI 树:text / content-desc 子串匹配(原生控件最快最准)
|
||||
for kw in ({"textContains": text}, {"descriptionContains": text}):
|
||||
try:
|
||||
el = d(**kw)
|
||||
if el.exists(timeout=1.5):
|
||||
b = el.bounds
|
||||
el.click()
|
||||
return jsonify({"ok": True, "found": True, "method": "ui",
|
||||
"matched": text,
|
||||
"x": (b[0] + b[2]) // 2, "y": (b[1] + b[3]) // 2})
|
||||
except Exception:
|
||||
continue
|
||||
# ② OCR 兜底:截图找文字中心(UI 树没有的渲染文字)
|
||||
img = d.screenshot()
|
||||
if img is not None:
|
||||
from core.ocr import find_on_screen
|
||||
found, center, matched = find_on_screen(img, text)
|
||||
if found:
|
||||
d.click(*center)
|
||||
return jsonify({"ok": True, "found": True, "method": "ocr",
|
||||
"matched": matched,
|
||||
"x": center[0], "y": center[1]})
|
||||
return jsonify({"ok": True, "found": False, "method": "",
|
||||
"matched": "", "x": 0, "y": 0})
|
||||
except Exception as e:
|
||||
_screen_invalidate(serial)
|
||||
return jsonify({"ok": False, "error": f"文字点击失败: {e}"}), 503
|
||||
|
||||
@bp.route("/api/screen/swipe", methods=["POST"])
|
||||
@perm_required(PERM_DEVICES)
|
||||
def api_screen_swipe():
|
||||
|
||||
Reference in New Issue
Block a user