fix: AI 点击精度优化——①坐标吸附:/api/screen/tap 加 snap=1,MCP de_tap 固定开启(dump UI 树找包含点击点的最小可点击元素点中心,模型坐标偏 20-50px 也点得准,大屏触控不带 snap 行为不变);②新 de_tap_text 语义点击:按屏幕可见文字一次完成找+点(UI 树 textContains/descriptionContains → OCR 中心兜底,WebView/图片文字也能点);③de_tap_element 支持 text_contains/desc_contains 模糊匹配;④de_ui_tree 可点击元素优先 + limit 参数防 token 膨胀;⑤agent 提示词重写:文字语义点击优先、坐标仅纯图形兜底且自动吸附、点击后无变化禁止重复同坐标
This commit is contained in:
+98
-3
@@ -1,6 +1,7 @@
|
||||
"""监控域 API:状态/运行控制/设备操作/远程看屏。"""
|
||||
import time
|
||||
import threading
|
||||
import re
|
||||
import subprocess
|
||||
import shlex
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
@@ -476,21 +477,115 @@ def api_screen_thumb():
|
||||
return jsonify({"ok": False, "error": str(e)}), 503
|
||||
|
||||
|
||||
def _snap_to_clickable(d, x, y):
|
||||
"""坐标吸附:找包含 (x,y) 的最小可点击元素,返回其 bounds 中心。
|
||||
|
||||
AI/触控给的坐标常偏离目标 20-50px(模型视觉定位误差)。dump 当前 UI 树后
|
||||
遍历 clickable 节点:点击点落在哪个可点元素内就点它的中心——偏了也点得准;
|
||||
取包含元素中面积最小者(最具体的那个)。点空白处(收键盘等)无包含元素则
|
||||
原坐标返回。dump 失败也不阻塞,直接原坐标。返回 (cx, cy, snapped, label)。
|
||||
"""
|
||||
try:
|
||||
import xml.etree.ElementTree as ET
|
||||
xml_str = d.dump_hierarchy()
|
||||
best = None # (area, cx, cy, label)
|
||||
for node in ET.fromstring(xml_str).iter("node"):
|
||||
if node.get("clickable") != "true":
|
||||
continue
|
||||
if node.get("enabled") == "false":
|
||||
continue
|
||||
m = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", node.get("bounds", ""))
|
||||
if not m:
|
||||
continue
|
||||
x1, y1, x2, y2 = map(int, m.groups())
|
||||
if x1 <= x <= x2 and y1 <= y <= y2:
|
||||
area = (x2 - x1) * (y2 - y1)
|
||||
if best is None or area < best[0]:
|
||||
best = (area, (x1 + x2) // 2, (y1 + y2) // 2,
|
||||
(node.get("text") or node.get("content-desc") or "")[:40])
|
||||
if best:
|
||||
return best[1], best[2], True, best[3]
|
||||
except Exception:
|
||||
pass
|
||||
return x, y, False, ""
|
||||
|
||||
|
||||
@bp.route("/api/screen/tap", methods=["POST"])
|
||||
@perm_required(PERM_DEVICES)
|
||||
def api_screen_tap():
|
||||
"""点击:{serial, x, y}(设备原生分辨率坐标)。"""
|
||||
"""点击:{serial, x, y, snap?}(设备原生分辨率坐标)。
|
||||
|
||||
snap=1 时先吸附:点落在可点击元素内则改点元素中心(MCP/AI 场景用,粗略
|
||||
坐标也能点准);大屏精确触控不带 snap 保持原行为。
|
||||
返回 {ok, snapped, x, y, label}。
|
||||
"""
|
||||
data = request.json or {}
|
||||
serial, x, y = data.get("serial", ""), data.get("x"), data.get("y")
|
||||
snap = int(data.get("snap", 0) or 0) == 1
|
||||
if not serial or x is None or y is None:
|
||||
return jsonify({"ok": False, "error": "缺少 serial/x/y"}), 400
|
||||
try:
|
||||
_screen_get_device(serial).click(int(x), int(y))
|
||||
return jsonify({"ok": True})
|
||||
d = _screen_get_device(serial)
|
||||
sx, sy = int(x), int(y)
|
||||
label = ""
|
||||
snapped = False
|
||||
if snap:
|
||||
sx, sy, snapped, label = _snap_to_clickable(d, sx, sy)
|
||||
d.click(sx, sy)
|
||||
return jsonify({"ok": True, "snapped": snapped, "x": sx, "y": sy,
|
||||
"label": label})
|
||||
except Exception as e:
|
||||
_screen_invalidate(serial)
|
||||
return jsonify({"ok": False, "error": f"点击失败: {e}"}), 503
|
||||
|
||||
|
||||
@bp.route("/api/screen/tap_text", methods=["POST"])
|
||||
@perm_required(PERM_DEVICES)
|
||||
def api_screen_tap_text():
|
||||
"""按屏幕文字点击:{serial, text}——找到显示该文字的位置并点中心(子串匹配)。
|
||||
|
||||
两步:① UI 树 textContains/descriptionContains 命中 → 点元素中心(原生控件);
|
||||
② 未命中 → 截图 OCR 找文字中心(WebView/图片/画布渲染的文字)。
|
||||
返回 {ok, found, method: ui|ocr, matched, x, y};found=false 表示屏幕确实
|
||||
没有该文字(业务结果非设备错误);OCR 不可用且 UI 未命中时也返回 found=false。
|
||||
"""
|
||||
data = request.json or {}
|
||||
serial = (data.get("serial") or "").strip()
|
||||
text = (data.get("text") or "").strip()
|
||||
if not serial or not text:
|
||||
return jsonify({"ok": False, "error": "缺少 serial/text"}), 400
|
||||
if len(text) > 100:
|
||||
return jsonify({"ok": False, "error": "文字过长(≤100 字符)"}), 400
|
||||
try:
|
||||
d = _screen_get_device(serial)
|
||||
# ① UI 树:text / content-desc 子串匹配(原生控件最快最准)
|
||||
for kw in ({"textContains": text}, {"descriptionContains": text}):
|
||||
try:
|
||||
el = d(**kw)
|
||||
if el.exists(timeout=1.5):
|
||||
b = el.bounds
|
||||
el.click()
|
||||
return jsonify({"ok": True, "found": True, "method": "ui",
|
||||
"matched": text,
|
||||
"x": (b[0] + b[2]) // 2, "y": (b[1] + b[3]) // 2})
|
||||
except Exception:
|
||||
continue
|
||||
# ② OCR 兜底:截图找文字中心(UI 树没有的渲染文字)
|
||||
img = d.screenshot()
|
||||
if img is not None:
|
||||
from core.ocr import find_on_screen
|
||||
found, center, matched = find_on_screen(img, text)
|
||||
if found:
|
||||
d.click(*center)
|
||||
return jsonify({"ok": True, "found": True, "method": "ocr",
|
||||
"matched": matched,
|
||||
"x": center[0], "y": center[1]})
|
||||
return jsonify({"ok": True, "found": False, "method": "",
|
||||
"matched": "", "x": 0, "y": 0})
|
||||
except Exception as e:
|
||||
_screen_invalidate(serial)
|
||||
return jsonify({"ok": False, "error": f"文字点击失败: {e}"}), 503
|
||||
|
||||
@bp.route("/api/screen/swipe", methods=["POST"])
|
||||
@perm_required(PERM_DEVICES)
|
||||
def api_screen_swipe():
|
||||
|
||||
Reference in New Issue
Block a user