From 972db13f818a483bd7066a5da3dea50383d0725f Mon Sep 17 00:00:00 2001 From: butubb <1422726308@qq.com> Date: Thu, 10 Sep 2026 15:31:39 +0800 Subject: [PATCH] =?UTF-8?q?fix:=20=E7=BB=8F=E9=AA=8C/=E5=8A=A8=E4=BD=9C?= =?UTF-8?q?=E8=92=B8=E9=A6=8F=E4=B8=8D=E5=86=8D=E9=9D=99=E9=BB=98=E4=B8=A2?= =?UTF-8?q?=E5=BC=83=EF=BC=88=E5=85=B3=E6=8E=A8=E7=90=86=20+=20=E8=B4=A8?= =?UTF-8?q?=E9=87=8F=E9=97=A8=E6=A7=9B=20+=20=E6=88=AA=E6=96=AD=E5=AE=B9?= =?UTF-8?q?=E5=BF=8D=EF=BC=89+=20=E5=89=8D=E7=AB=AF=E4=BC=9A=E8=AF=9D?= =?UTF-8?q?=E6=98=BE=E7=A4=BA=20ID?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 问题:蒸馏模型把 token 预算烧在 reasoning 上 → content 为空/被截断(finish_reason=length), 代码只读 content → 经验与动作被静默丢弃("使用小红书找苏州饭店"跑完什么都没存,动作侧日志 '原始输出 0 字符')。 修复(web/agent_api.py): - 蒸馏调用统一加 "thinking": {"type":"disabled"}(该代理支持;实测关掉后 reasoning=0、 配方 3/3 合格)——关键修复 - 配方:纯文本问法 + _recipe_ok 质量门槛(过短/含省略号占位丢弃,避免把提示词示例当真配方; 曾因提示词里写了占位示例,模型照抄成 "1. …\n2. …" 存进库)+ 空则重试一次; 仅动作提炼允许回退 reasoning_content(配方不回退,防思考草稿污染) - 动作:JSON 输出 + _loads_lenient 截断容忍(逐对象抢救)+ {action,params} 形状归一 + 输入/产出限量(≤10 步、≤3 动作×4 步);失败日志带样本 - 前端 agent.js:会话列表显示会话 ID 前 8 位(等宽小字),点击复制完整 ID(便于引用 conv=) - doc/ARCHITECTURE.md:§3.8 序号语义 + §5.4 蒸馏健壮性与会话 ID 说明 实测:真机复跑同一句需求 → 经验已保存(配方 196 字符)+ 动作经验已保存 2 条 --- doc/ARCHITECTURE.md | 15 ++++++ static/admin/agent.js | 23 +++++++- web/agent_api.py | 119 ++++++++++++++++++++++++++++++++++-------- 3 files changed, 135 insertions(+), 22 deletions(-) diff --git a/doc/ARCHITECTURE.md b/doc/ARCHITECTURE.md index 5942a3a..68798fd 100644 --- a/doc/ARCHITECTURE.md +++ b/doc/ARCHITECTURE.md @@ -230,6 +230,11 @@ APK 上传/解析/批量安装。 元素树解析:递归提取每个节点的 `resource-id/text/content-desc/class/bounds` 等属性,并推荐最佳选择器(优先 xpath)。 +**XPath 序号语义(重要)**:同一属性多个实例时,生成 **`(//*[@resource-id="x"])[k]`**(整体加括号 = 第 k 个匹配)。 +不可写成 `//*[@resource-id="x"][k]`——那在 XPath 里是"**在其父节点中排第 k**",多实例时 `[2..n]` 会全部匹配不到 +(2026-09-10 实测修复:抖音底部 4 个 tab 同 id,旧写法除 `[1]` 外全失效)。执行器 `tasks/generic/task.py` +对**历史遗留**的 `//*[@attr=…][k]` 形态做窄范围纠正(`_norm_legacy_xpath`,只改前缀、不动结构路径的兄弟序号)。 + ### 3.9 屏幕 OCR(`core/ocr.py`) 条件判断的 `ocr` 选择器实现:截屏 → RapidOCR(ONNX 推理,中英文模型随包内置)→ 关键词匹配 → 返回文字中心像素坐标(与 u2 `d.click` 一致)。 @@ -339,11 +344,21 @@ tasks/douyin/actions/like.py — @register_action(ACTIONS) LikeAction AI 控制台(顶级 Tab)右上角两个模态框,管理自进化记忆: - **🧠 经验库**:任务级经验(`agent_experience`,整任务配方)+ 每日 AI 巡检建议(删除需人工确认)。 - **🎬 动作库**:动作级经验(`agent_action`)——命名动作(可含 1~N 步)+ 编辑器 schema 步骤 + **元素定位(禁坐标)**;由任务成功后从**成功步骤**自动蒸馏,执行前按名/别名召回注入;面板支持查看/编辑/删除/手动新建(保存经服务端校验,坐标步骤被拒)。 +- **会话列表显示会话 ID**(前 8 位,等宽小字),点击即复制完整 ID——便于反馈问题时引用 `conv=`。 + +> **蒸馏健壮性(2026-09-10)**:经验/动作靠**模型蒸馏**落库。推理型模型会把 token 预算烧在 `reasoning` 上,导致 `content` 为空或被截断(`finish_reason=length`)→ 早期只读 `content`,经验/动作被**静默丢弃**("小红书·苏州饭店"案例)。现策略: +> 1. 蒸馏调用**关闭推理**:`"thinking": {"type": "disabled"}`(该代理支持;实测关掉后 reasoning=0、正文正常,配方 3/3 合格)——这是关键修复; +> 2. 配方用**纯文本问法**(不要放可照抄的占位示例,否则模型会原样当配方存下来)+ 质量门槛 `_recipe_ok`(过短/含省略号占位 → 丢弃并重试); +> 3. 动作提炼用 JSON + **截断容忍**提取(`_loads_lenient` 逐对象抢救)+ 顶层 `{action,params}` 形状归一 + 输入/产出限量(≤10 步输入、≤3 动作×4 步); +> 4. 两类失败都有日志(`经验提炼:` / `动作提炼:` 含样本),不再静默。 ### 5.5 元素抓取模态框 独立的第二层模态框(`el-picker-overlay`,z-index 1100),不影响任务编辑窗口: 1. 选择设备 → 2. 加载截图 + 元素树 → 3. 点击元素/边界框 → 4. 回填选择器 +5. **抓取时直接验证**(每条元素右侧两个按钮,不会与"点击回填"冲突): + - 「▶ 点一下」:按元素 `bounds` 中心在设备上真点一次(`POST /api/screen/tap`,`snap=1` 自动吸附到可点元素),返回吸附结果并自动刷新截图——用于确认位置/是否可达; + - 「✓ 测选择器」:用**将填入的选择器**真跑一次 click(`POST /api/steps/test`),返回 `命中/未找到/已执行`——用于确认回填的选择器在真实界面能命中(元素无有效选择器时不显示此按钮)。 --- diff --git a/static/admin/agent.js b/static/admin/agent.js index a420bfc..21d02e4 100644 --- a/static/admin/agent.js +++ b/static/admin/agent.js @@ -252,15 +252,36 @@ function renderConvList(){ } list.innerHTML = convs.map(c=>{ const t = c.title || '新会话'; + const sid = String(c.id || ''); return '
' +'' +'
'+esc(t)+'
' - +'
'+esc(c.updated_at||'')+(c.count ? ' · '+c.count+' 轮' : '')+'
'; + +'
'+esc(c.updated_at||'')+(c.count ? ' · '+c.count+' 轮' : '') + +' · #'+esc(sid.slice(0,8))+'' + +'
'; }).join(''); }); } +// 复制会话 ID(便于反馈问题时引用,如 conv=2ba4da0e43) +function copyConvId(id){ + const done = ()=>showToast('会话 ID 已复制: '+id,'success'); + try{ + if(navigator.clipboard && navigator.clipboard.writeText){ + navigator.clipboard.writeText(id).then(done).catch(()=>fallback()); + return; + } + }catch(e){} + fallback(); + function fallback(){ + const ta=document.createElement('textarea'); + ta.value=id; document.body.appendChild(ta); ta.select(); + try{ document.execCommand('copy'); done(); }catch(e){ showToast('会话 ID: '+id,'success'); } + ta.remove(); + } +} function loadCurrentConvMessages(){ const chat = document.getElementById('agent-chat'); if(!chat) return; diff --git a/web/agent_api.py b/web/agent_api.py index 5096059..5dbeecd 100644 --- a/web/agent_api.py +++ b/web/agent_api.py @@ -226,6 +226,60 @@ def _find_experiences(prompt, limit=2, threshold=0.10): return "\n".join(parts), picked +def _msg_text(message, allow_reasoning=False): + """取模型回复正文。 + + 推理模型(DeepSeek 等)**偶发**把 token 预算全花在 reasoning 上、content 为空 + (finish_reason=length),直接读 content 会当空处理 → 经验/动作被静默丢弃 + (2026-09-10 实测)。默认**不回退 reasoning**(那是思考草稿,做"配方"会污染); + 仅在能结构化解析的场景(动作 JSON 提取)才允许回退。 + """ + m = message or {} + text = (m.get("content") or "").strip() + if text: + return text + if allow_reasoning: + return (m.get("reasoning_content") or "").strip() + return "" + + +_RECIPE_JSON_RE = re.compile(r'"recipe"\s*:\s*"((?:[^"\\]|\\.)*)"', re.S) + + +def _recipe_ok(recipe): + """配方质量门槛:太短、或只有序号+省略号(模型照抄占位)视为无效。 + + 2026-09-10 实测:提示词里写了示例 JSON,模型会把占位符 `1. …\\n2. …` + 原样当成配方存下来 → 这里拦掉,宁可重试/不存,也不污染经验库。 + """ + t = (recipe or "").strip() + if len(t) < 15: + return False + if "…" in t and len(t) < 30: + return False + return True + + +def _extract_recipe(text): + """从(可能被截断的)模型输出里取配方正文。 + + 蒸馏提示词要求输出 {"recipe": "…"};推理模型会把预算烧在 reasoning 上并 + 截断(finish_reason=length),严格 json.loads 会失败,故用正则直接抠字段。 + """ + if not text: + return "" + m = _RECIPE_JSON_RE.search(text) + if m: + try: + return json.loads('"' + m.group(1) + '"').strip() + except Exception: + return m.group(1).replace("\\n", "\n").strip() + t = text.strip() + if t.startswith("{") or t.startswith("["): + return "" # JSON 外壳但没抠到 recipe:视为无效,避免把草稿当配方 + return t + + def _clean_recipe(recipe): """清洗蒸馏输出:去掉模型可能加的「操作配方:」之类前缀标签与包裹引号。 @@ -247,22 +301,34 @@ def _distill_experience(cfg, prompt, tool_seq): import httpx body = { "model": cfg.get("model") or "deepseek-v4-flash-vision-exp", + # 关推理:蒸馏是"给定轨迹写配方"的确定性任务,开启 thinking 会占满 + # 预算导致 content 空/被截断(实测);该代理支持 thinking.type=disabled + "thinking": {"type": "disabled"}, "messages": [{"role": "user", "content": "以下是一次成功的手机自动化操作记录。请提炼成简洁的" "「操作配方」(2-6 步,每步:目标 → 用哪个工具)," - "供下次同类任务参考。不要解释,直接输出配方。\n" + "供下次同类任务参考。不要解释,直接输出配方正文。\n" f"任务:{prompt[:300]}\n操作序列:{tool_seq[:800]}"}], - "max_tokens": 600, + "max_tokens": 800, } headers = {"Authorization": f"Bearer {cfg.get('api_key', '')}", "Content-Type": "application/json"} - r = httpx.post(f"{(cfg.get('api_base') or 'https://api.deepseek.com').rstrip('/')}/chat/completions", - json=body, headers=headers, timeout=25) - if r.status_code != 200: - return "" - j = r.json() - recipe = ((j.get("choices") or [{}])[0].get("message") or {}).get("content") or "" - return recipe.strip()[:1500] + url = f"{(cfg.get('api_base') or 'https://api.deepseek.com').rstrip('/')}/chat/completions" + # content 为空(推理吃满预算)时重试一次——实测同样提示第二次常能出正文 + for _attempt in (1, 2): + r = httpx.post(url, json=body, headers=headers, timeout=25) + if r.status_code != 200: + _log.warning(f"经验提炼: 模型返回 HTTP {r.status_code}") + continue + msg = (r.json().get("choices") or [{}])[0].get("message") or {} + # 正文优先;为空时从 reasoning 里抠(可能含清晰步骤),都要过质量门槛 + for cand in (_msg_text(msg), _extract_recipe(msg.get("reasoning_content"))): + if _recipe_ok(cand): + return cand.strip()[:1500] + if cand: + _log.info(f"经验提炼: 候选配方质量不足({len(cand)} 字符)被丢弃") + _log.info("经验提炼: 两次均未产出可用配方(推理占满预算/输出被截断/仅占位)") + return "" except Exception as e: _log.warning(f"经验提炼失败: {e}") return "" @@ -505,10 +571,13 @@ def _distill_actions(cfg, prompt, trace): _log.info(f"动作提炼: 轨迹 {len(trace or [])} 步, 成功可沉淀 {len(ok_ops)} 步") if not ok_ops: return [] + # 输入截断到前 10 步:步数过多会让模型输出变长被截断(实测 10 步 → 2002 字符无有效 JSON) + ok_ops = ok_ops[:10] lines = [f"- {t['tool']} 参数={t['args']} 结果={_brief_result(t['result'])}" for t in ok_ops] instruction = ( - "以下是一次成功的手机自动化操作的**成功步骤**。请把它们提炼为若干「动作」" - "(每个动作 = 一个有语义名的可复用单元,可含 1~N 步)。只输出 JSON 数组,不要解释。\n" + "以下是一次成功的手机自动化操作的**成功步骤**。请提炼为**至多 3 个**「动作」" + "(每个动作 = 一个有语义名的可复用单元,**至多 4 步**,字段尽量简短)。" + "只输出 JSON 数组,不要解释,不要思考过程。\n" "字段:name(动作名,如「打开抖音」「搜索关键词」);app(包名,未知则空串);" "aliases(别名数组);params(参数名数组,如[\"关键词\"]);steps(步骤数组)。\n" "steps 每步:type + params,type 取值:open_app{package} / click{selector_type," @@ -526,20 +595,28 @@ def _distill_actions(cfg, prompt, trace): try: import httpx body = {"model": cfg.get("model") or "deepseek-v4-flash-vision-exp", + "thinking": {"type": "disabled"}, # 同 _distill_experience:关推理 "messages": [{"role": "user", "content": instruction}], "max_tokens": 900} headers = {"Authorization": f"Bearer {cfg.get('api_key', '')}", "Content-Type": "application/json"} - r = httpx.post( - f"{(cfg.get('api_base') or 'https://api.deepseek.com').rstrip('/')}/chat/completions", - json=body, headers=headers, timeout=30) - if r.status_code != 200: - _log.warning(f"动作提炼: 模型返回 HTTP {r.status_code}") - return [] - content = (((r.json().get("choices") or [{}])[0].get("message") or {}).get("content") or "") + url = f"{(cfg.get('api_base') or 'https://api.deepseek.com').rstrip('/')}/chat/completions" + content = "" + # content 空(推理吃满预算)→ 重试一次;仍空则回退 reasoning_content + # (这里能结构化提取 JSON 数组,草稿里也常含可用 JSON,故允许回退) + for _attempt in (1, 2): + r = httpx.post(url, json=body, headers=headers, timeout=30) + if r.status_code != 200: + _log.warning(f"动作提炼: 模型返回 HTTP {r.status_code}") + continue + msg = (r.json().get("choices") or [{}])[0].get("message") or {} + content = _msg_text(msg, allow_reasoning=True) + if content: + break acts = _sanitize_actions(_loads_lenient(content)) if not acts: - _log.info(f"动作提炼: 解析后无有效动作(原始输出 {len(content)} 字符)") + _log.info(f"动作提炼: 解析后无有效动作(原始输出 {len(content)} 字符)" + f" 样本={content[:160]!r}") return acts except Exception as e: _log.warning(f"动作提炼失败: {e}") @@ -681,8 +758,8 @@ def _review_one_exp(cfg, prompt, recipe, tool_seq, hits): json=body, headers=headers, timeout=20) if r.status_code != 200: raise RuntimeError(f"HTTP {r.status_code}") - text = (((r.json() or {}).get("choices") or [{}])[0] - .get("message") or {}).get("content") or "" + text = _msg_text(((r.json() or {}).get("choices") or [{}])[0] + .get("message") or {}) m = re.search(r"\{.*\}", text, re.S) if not m: raise RuntimeError("响应无 JSON")