"""uiauto2 (uiautodev) 本地服务客户端封装。 uiautodev 启动后本地监听 http://localhost:20242,提供设备元素树查询。 前端步骤编辑器"抓取元素"按钮通过本模块拉取当前设备 UI 树供选择。 依赖: - 用户需先 `pip install uiautodev` 并运行 `uiauto.dev` 启动本地服务 - 未运行时本模块函数返回友好错误,不抛异常 API 参考(uiautodev 0.14): - GET /api/info — 服务信息(用于探测是否运行) - GET /api/android/{serial}/dump_hierarchy — 元素树 JSON """ import requests from collections import Counter, defaultdict from core.logger import get_logger _log = get_logger("core.uiauto") # uiautodev 默认本地端口(固定 20242) _UIAUTO_BASE = "http://localhost:20242" # 请求超时(秒)。connect 超时短,避免前端等太久 _TIMEOUT = (1, 8) def is_running(): """探测 uiauto2 本地服务是否在运行。""" try: r = requests.get(f"{_UIAUTO_BASE}/api/info", timeout=_TIMEOUT) return r.status_code == 200 except Exception: return False def list_devices(): """获取 uiauto2 已连接的设备列表。 返回 (ok, data_or_error): ok=True — data 是设备列表 [{serial, model, product, name, status}, ...] ok=False — data 是错误消息字符串 """ try: r = requests.get(f"{_UIAUTO_BASE}/api/android/list", timeout=_TIMEOUT) r.raise_for_status() data = r.json() return True, data except requests.exceptions.ConnectionError: return False, "uiauto2 未启动,请运行 `uiauto.dev`" except Exception as e: _log.warning(f"list_devices 异常: {e}") return False, f"获取设备列表失败: {e}" def get_screenshot(serial): """获取设备截图(通过 uiauto2 服务)。 返回 (ok, data_or_error): ok=True — data 是 JPEG 二进制数据 ok=False — data 是错误消息字符串 """ if not serial: return False, "缺少 serial" try: r = requests.get( f"{_UIAUTO_BASE}/api/android/{serial}/screenshot/0", timeout=_TIMEOUT, ) r.raise_for_status() return True, r.content except requests.exceptions.ConnectionError: return False, "uiauto2 未启动" except Exception as e: _log.warning(f"get_screenshot 异常: {e}") return False, f"截图失败: {e}" def get_elements(serial): """获取指定设备的 UI 元素树。 返回 (ok, data_or_error): ok=True — data 是元素列表 [{name, attrs..., suggested:{type,value,semantic?,via?,indexed?,occ?,total?,broad?}}, ...] ok=False — data 是错误消息字符串 元素树由 uiautodev 的 dump_hierarchy 返回(JSON)。我们递归提取每个节点的 关键属性(resource-id/text/description/class/bounds...),供前端列表展示和选择。 选择器精度策略(精确到具体按钮的关键,优先级见 doc/research/U2_ELEMENT_SELECTORS.md §三): 1. 有 resource-id/text/content-desc 的元素: 先预统计该属性在整棵树中的出现次数—— - 唯一出现:直接用属性选择器 //*[@resource-id="x"] - 重复出现(如抖音底部导航的 tab 同 id,个数还随灰度版本变): **优先用第二个属性把目标单独圈出来**(语义选择器): //*[@resource-id="x" and @text="我"] 它只认"这个元素的文字/id 是什么",不认"同 id 一共几个", 换设备/换版本依然成立 → 标记 semantic + via(用于限定的属性)。 - 两个属性组合仍圈不出来时,才退回**整体加括号**的 `(//*[@resource-id="x"])[k]`(标记 indexed,前端醒目提示脆弱)。 注意 XPath 语义:`//*[@id="x"][k]` 是"在其父节点中排第 k",不是 第 k 个匹配——历史实现踩过这个坑(多实例时 [2..n] 全部失配)。 2. 无任何属性的元素: 用最近一个有属性祖先的选择器限定范围 + 同 class 兄弟序号定位 (如 //*[@resource-id="x"]/*[@class="android.widget.ImageView"][2]); 整棵树都没有属性时退化为从根开始的结构路径 (//hierarchy/*[1]/*[3]…,标记 broad,前端提示脆弱)。 注意:dump 的 XML 标签一律是 ,class 在 @class 上 —— 任何把 class 名当标签名的写法(//FrameLayout[1])都永远匹配不到,别再用。 """ if not serial: return False, "缺少 serial" try: r = requests.get( f"{_UIAUTO_BASE}/api/android/{serial}/hierarchy", timeout=_TIMEOUT, ) if r.status_code == 404: return False, f"设备 {serial} 未连接到 uiauto2" r.raise_for_status() data = r.json() # uiautodev hierarchy 返回 Node 树:{key, name, bounds, rect, properties, children} # 提取成扁平的可选列表(保留层级缩进信息) elements = [] _extract(data, elements) if not elements: return False, "当前界面未抓取到元素" return True, elements except requests.exceptions.ConnectionError: return False, "uiauto2 未启动,请运行 `uiauto.dev`" except requests.exceptions.Timeout: return False, "uiauto2 请求超时" except Exception as e: _log.warning(f"get_elements 异常: {e}") return False, f"抓取失败: {e}" def _child_path(path): """内部路径 `/0/2/1` → XPath 段 `/*[1]/*[3]/*[2]`(按**子节点位置**逐层定位)。 用于"整棵树都没有可用属性"时的兜底结构路径。为什么按位置而不是 `@index`: 实测 Android dump 里同级节点的 index 属性**会重复**(状态栏/内容区/导航栏 三个兄弟的 index 全是 "0"),拿它定位会一次命中好几个。 """ return "".join(f"/*[{int(seg) + 1}]" for seg in path.strip("/").split("/") if seg) def _xpath(conds): """由条件列表拼 XPath:[(attr, value), ...] → `//*[@a="1" and @b="2"]`。 单个条件即 `//*[@a="1"]`。多个条件是**并列且**,任一条不成立就不匹配。 """ return "//*[" + " and ".join(f"@{a}={_xpath_q(v)}" for a, v in conds) + "]" def _xpath_q(v): """XPath 字符串字面量:优先双引号,值含双引号时改用单引号包裹(XPath 1.0 无转义)。""" if '"' in v: return "'" + v + "'" return '"' + v + '"' def _extract(root, out): """把 uiautodev 元素树扁平化为可选列表,并为每个节点生成精确选择器建议。""" # 全树属性出现次数(预统计:判断"这个值重不重复",决定要不要消歧) id_cnt, text_cnt, desc_cnt = Counter(), Counter(), Counter() # 同属性值 → 该值对应的全部节点 properties(语义消歧要在"同值节点"里比第二个属性) id_nodes, text_nodes, desc_nodes = defaultdict(list), defaultdict(list), defaultdict(list) # 文档顺序已出现次数(决定当前元素是第几个) seen_id, seen_text, seen_desc = Counter(), Counter(), Counter() def count_attrs(node): if not isinstance(node, dict): return props = node.get("properties") or {} rid, text, desc = (props.get(k, "") for k in ("resource-id", "text", "content-desc")) if rid: id_cnt[rid] += 1 id_nodes[rid].append(props) if text: text_cnt[text] += 1 text_nodes[text].append(props) if desc: desc_cnt[desc] += 1 desc_nodes[desc].append(props) for c in node.get("children") or []: count_attrs(c) # 消歧时可用的次要属性(按"越稳越靠前"排:文字 > 描述 > class) _SECONDARY_ATTRS = ("text", "content-desc", "class") def secondary_conds(peers, props): """属性值重复时,找一组能**把本节点单独圈出来**的次要属性。 例:抖音底部 4 个 tab 共享 resource-id `…:0qf`,但 text 分别是 首页/朋友/消息/我 —— 于是 //*[@resource-id="…:0qf" and @text="我"] 这种**语义选择器**只依赖"目标自己长什么样",不依赖"同 id 一共几个": 灰度版把 tab 从 4 个变 3 个、换一台设备,它照样命中。 对比 `(…)[k]`:那是"第 k 个匹配",界面一变就指到别的元素上 → 点不中。 返回 ([(attr, value), ...], via) 或 (None, None)。peers 是"同主属性值"的 全部节点 properties —— 唯一性只需在这一集合内成立(任何匹配都必然在此集合中)。 """ def matched(conds): return sum(1 for p in peers if all((p.get(a) or "") == v for a, v in conds)) usable = [(a, props.get(a) or "") for a in _SECONDARY_ATTRS] usable = [(a, v) for a, v in usable if v.strip()] # 空值不能做限定条件 # 单个次要属性够了就用它(选择器最短) for a, v in usable: if matched([(a, v)]) == 1: return [(a, v)], a # 单个不够 → 两个次要属性组合(如 @text + @class) for i in range(len(usable)): for j in range(i + 1, len(usable)): conds = [usable[i], usable[j]] if matched(conds) == 1: return conds, f"{usable[i][0]}+{usable[j][0]}" return None, None def attr_selector(attr, value, cnt, seen, peers, props): """属性选择器:唯一直接出;重复时**先语义消歧,再退化到序号**。 语义消歧(2026-09-13 新增,见 doc/research/U2_ELEMENT_SELECTORS.md §五): //*[@resource-id="x" and @text="我"] ← 不依赖元素个数,换设备/换版本仍成立 退而求其次(同 id 同文字都分不开,如列表里的重复项): (//*[@resource-id="x"])[2] ← 精确到第 2 个匹配,但界面一变就失配 重要(XPath 位置谓词语义): //*[@resource-id="x"][2] → 「在**其父节点**中排第 2 的属性匹配」,**不是**第 2 个匹配 (//*[@resource-id="x"])[2] → 「第 2 个匹配」← 我们要的 历史实现写成前者,导致同 id 多实例(如抖音底部导航 4 个 tab 同 id)时 [2..n] 全部匹配不到 → 运行时"未找到元素"(2026-09-10 实测修复)。 """ seen[value] += 1 conds = [(attr, value)] if cnt[value] > 1: extra, via = secondary_conds(peers, props) if extra: conds += extra else: val = f"({_xpath(conds)})[{seen[value]}]" return val, {"type": "xpath", "value": val, "indexed": True, "occ": seen[value], "total": cnt[value]} val = _xpath(conds) info = {"type": "xpath", "value": val} if len(conds) > 1: info.update(semantic=True, via=via) return val, info def flatten(node, depth=0, path="", ctx=None, tag_path="", tag_index=1): """递归扁平化。 ctx: 最近一个有属性祖先的选择器(含序号),无属性元素用它限定范围。 tag_path: 从根到本节点的完整结构路径(class + 同class兄弟序号),兜底用。 tag_index: 本节点在父下同 class 兄弟中的序号(1-based)。 """ if not isinstance(node, dict): return props = node.get("properties") or {} name = node.get("name") or props.get("class") or "" # bounds:优先取 properties 中的原始字符串 "[x1,y1][x2,y2]"(前端渲染 overlay 用) bounds_str = props.get("bounds", "") if not bounds_str and node.get("rect"): # rect 兜底:只有整数像素坐标才用;uiautodev 归一化浮点坐标无法换算像素 r = node["rect"] try: vals = [r["x"], r["y"], r["x"] + r["width"], r["y"] + r["height"]] if all(isinstance(v, (int, float)) and float(v).is_integer() for v in vals): bounds_str = f"[{int(vals[0])},{int(vals[1])}][{int(vals[2])},{int(vals[3])}]" except (TypeError, KeyError): pass rid = props.get("resource-id", "") text = props.get("text", "") desc = props.get("content-desc", "") cls = props.get("class", "") tag = cls.split(".")[-1] if cls else "*" # 始终带同 class 兄弟序号(含 [1]),结构路径才精确无歧义; # class 为空的层(tag=*)不参与结构路径,避免 //*[N] 前缀污染导致选择器定位到任意节点 if tag != "*": # 步进必须写成 `*[@class="…"]`,**不能**写 `FrameLayout[1]`: # Android dump 的 XML 里每个元素的**标签名都是 ``**,class 在 # @class 属性上 —— 写成标签名会永远匹配不到任何东西(见下方 broad 注释)。 seg = f"*[@class={_xpath_q(cls)}][{tag_index}]" own_tag_path = f"{tag_path}/{seg}" if tag_path else seg else: seg = "" own_tag_path = tag_path # ---- 推荐选择器:唯一属性 > 锚点祖先限定 > 全结构路径 ---- if rid: child_ctx, suggested = attr_selector("resource-id", rid, id_cnt, seen_id, id_nodes[rid], props) elif text: child_ctx, suggested = attr_selector("text", text, text_cnt, seen_text, text_nodes[text], props) elif desc: child_ctx, suggested = attr_selector("content-desc", desc, desc_cnt, seen_desc, desc_nodes[desc], props) elif ctx: val = f"{ctx}/{seg}" suggested = {"type": "xpath", "value": val} child_ctx = val else: if tag == "*": # class 也为空的元素:结构路径 //*[N] 无意义(匹配到文档任意节点), # 标记 invalid,前端提示不可选,避免回填垃圾选择器导致"点击不到" suggested = {"type": "xpath", "value": "", "invalid": True, "reason": "该元素无可用属性(id/文本/class),无法生成可靠选择器"} else: # 无唯一属性且无锚点祖先:用从根开始的结构路径(脆弱,标记 broad 让前端提示)。 # 按子节点位置逐层定位 —— 历史实现写成 //FrameLayout[1]/LinearLayout[2] # 这种"class 当标签名"的路径,**永远零命中**(dump 的标签全是 ); # 实测 248 个元素里 26 条死选择器,2026-09-13 修复。 suggested = {"type": "xpath", "value": f"//hierarchy{_child_path(path)}", "broad": True} child_ctx = None out.append({ "depth": depth, "path": path, "name": name, "resource_id": rid, "text": text, "description": desc, "class": cls, "package": props.get("package", ""), "clickable": props.get("clickable", ""), "bounds": bounds_str, "suggested": suggested, }) # 递归子节点,同时计算每个子节点在父下同 class 兄弟中的序号 children = node.get("children") or [] for i, child in enumerate(children): cprops = child.get("properties") or {} ccls = cprops.get("class", "") ctag = ccls.split(".")[-1] if ccls else "*" same = 1 + sum( 1 for prev in children[:i] if ((prev.get("properties") or {}).get("class", "")).split(".")[-1] == ctag ) flatten(child, depth + 1, f"{path}/{i}", child_ctx, own_tag_path, same) count_attrs(root) flatten(root) # ================== 原生 u2 快照(截图 + 元素树一次取齐) ================== def _xml_to_node(elem): """把 uiautomator2 的 XML 节点转成 uiautodev 那套 {name, properties, children}。 两边的属性名本来就一样(resource-id / text / content-desc / class / bounds 字符串), 所以只要套一层壳,就能原样复用上面的 _extract(选择器建议逻辑不用写第二遍)。 """ props = dict(elem.attrib) return {"name": props.get("class", ""), "properties": props, "children": [_xml_to_node(c) for c in list(elem)]} def _shots_differ(a, b, threshold=8): """两张截图是不是"明显不一样"(用来判断抓取期间界面有没有在动)。 逐像素求差的包围盒 → 用"变化区域占比"判断,避免个别像素抖动误报。 """ try: from PIL import ImageChops if a.size != b.size: return True diff = ImageChops.difference(a.convert("RGB"), b.convert("RGB")) bbox = diff.getbbox() if not bbox: return False area = (bbox[2] - bbox[0]) * (bbox[3] - bbox[1]) total = a.size[0] * a.size[1] # 变化面积超过阈值(默认 8%)才算"界面在动" return (area / total) * 100.0 > threshold except Exception: return False def snapshot(serial, quality=85): """一次取齐:设备截图 + 元素树(**同一个 u2 连接、背靠背取**)。 为什么不用现在的"两个接口":截图和元素树分两次取时,中间隔着 dump 本身的 1.3~1.8 秒;界面只要在动画(信息流/视频/加载),框就会落在旧位置上。 顺带做**双截图校验**:dump 前后各截一张,两张差得多就标 `unstable`, 让前端明确提示"界面在变化中,请停在静止界面再抓"——而不是悄悄给一个错位的框。 返回 (ok, 数据 | 错误信息)。数据形如: {"image": "data:image/jpeg;base64,…", "width": 720, "height": 1650, "elements": [...], "unstable": false, "screen_state": "on", "cost_ms": 1800} """ import base64 import io import time as _t import xml.etree.ElementTree as ET import uiautomator2 as u2 t0 = _t.time() try: d = u2.connect(serial) img1 = d.screenshot() # 先截:用户看到的就是这一刻 # 屏幕开关状态(MCP 的 de_snapshot 要用它判断"要不要先 de_wake"; # 本来只有 /api/screen/thumb 的响应头里有,这里顺手带上) try: screen_on = d.info.get("screenOn") except Exception: screen_on = None xml = d.dump_hierarchy() # 慢的一步(1.3~1.8s) img2 = d.screenshot() # 再截:和第一张比对 except Exception as e: return False, f"抓取失败: {e}" unstable = _shots_differ(img1, img2) if not unstable: img2.close() try: buf = io.BytesIO() img1.convert("RGB").save(buf, format="JPEG", quality=quality) image = "data:image/jpeg;base64," + base64.b64encode(buf.getvalue()).decode() w, h = img1.size except Exception as e: return False, f"截图编码失败: {e}" elements = [] try: root = ET.fromstring(xml) _extract(_xml_to_node(root), elements) except Exception as e: return False, f"元素树解析失败: {e}" return True, {"image": image, "width": w, "height": h, "elements": elements, "unstable": unstable, "screen_state": ("on" if screen_on else "off") if screen_on is not None else "unknown", "cost_ms": int((_t.time() - t0) * 1000)}