From 104964aa536fc919214e0d44fb65779f18b50db8 Mon Sep 17 00:00:00 2001 From: butubb <1422726308@qq.com> Date: Thu, 10 Sep 2026 13:53:04 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E5=8A=A8=E4=BD=9C=E7=BB=8F=E9=AA=8C?= =?UTF-8?q?=E5=BA=93(agent=5Faction)=E2=80=94=E2=80=94=E6=88=90=E5=8A=9F?= =?UTF-8?q?=E6=AD=A5=E9=AA=A4=E8=92=B8=E9=A6=8F=E5=91=BD=E5=90=8D=E5=8A=A8?= =?UTF-8?q?=E4=BD=9C(=E5=B8=A6=E5=85=83=E7=B4=A0=E5=AE=9A=E4=BD=8D/?= =?UTF-8?q?=E7=A6=81=E5=9D=90=E6=A0=87)+=20=E6=89=A7=E8=A1=8C=E5=89=8D?= =?UTF-8?q?=E5=8F=AC=E5=9B=9E=E6=B3=A8=E5=85=A5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 新表 agent_action(name/app/aliases/params/steps/preconditions/hits/时间), 独立于人工维护的 custom_action(2B 决策):AI 自学动作不污染手建动作 - 沉淀:任务成功后从**成功**工具轨迹(_ACTION_TOOLS: open_app/tap_text/tap_element/ type_text/clipboard/swipe/press_key/wake/sleep)用模型蒸馏为命名动作;steps 用 编辑器 schema,**必须元素定位**(xpath/text/resourceId/description…), **显式剔除 click_xy 等坐标类**;on_tool 记录带 result 的结构化轨迹以判成败 - 兼容模型形状漂移:顶层 {action,params} 自动归一为 {name,steps};宽容 JSON 解析 (围栏/尾逗号/中文引号/坏对象逐条抢救),实测模型常返回带语法错误的 JSON - 召回:执行前按动作名/别名命中(或相似度≥0.34)取 top3,注入 system prompt 「可复用动作」段(含元素定位),模型可跳过重新探索;hits 回写 - 文档同步:ARCHITECTURE §3.6(agent_action 表)、API.md(🧠 动作经验 伪卡片 + 动作库 说明)、AI_TASK_GEN P1(沉淀进展) 实测:跑「打开抖音,点搜索」→ 沉淀「打开抖音」;下一轮同指令命中并注入;日志 「命中可复用动作 1 个」「动作提炼: 轨迹 5 步, 成功可沉淀 1 步」「动作经验已保存 1 条」 --- doc/AI_TASK_GEN.md | 1 + doc/API.md | 4 +- doc/ARCHITECTURE.md | 3 +- web/agent_api.py | 386 +++++++++++++++++++++++++++++++++++++++++++- 4 files changed, 388 insertions(+), 6 deletions(-) diff --git a/doc/AI_TASK_GEN.md b/doc/AI_TASK_GEN.md index 80d9c62..178e107 100644 --- a/doc/AI_TASK_GEN.md +++ b/doc/AI_TASK_GEN.md @@ -136,6 +136,7 @@ ## 6. 里程碑 - **P0(本设计主体)**:designer 模式 → 自探(UI 树定位)→ 直接撰写 generic_steps draft → 服务端 schema 校验 → 前端步骤编辑器预填确认保存。验收:一句话在真实设备上生成一条可调度的 generic_steps,步骤全部来自 UI 树且编辑器可打开。 - **P1**:`douyin_nurture` 参数预设生成(照 default_params 结构);模板沉淀:把 draft+需求写入 `agent_experience`(新列存结构化 steps 或 JSON),相似需求注入参考;整链「演示试跑」(把 steps 在设备上以受控方式跑一遍并截图回报,需新增端点,复刻 `device_busy` 拒绝语义)。 + - **进展(2026-09-10)**:动作级沉淀已落地——`agent_action` 表 + 从**成功步骤**蒸馏"命名动作"(steps 用编辑器 schema、带元素定位、禁坐标)+ 执行前按名/别名召回注入(`web/agent_api.py`)。AI 建任务可直接把这些动作当作 generic_steps 的**预制件**复用。 - **P2**:自定义动作支持(内联展开成 group,或新增 `action_ref` 节点 + 执行器/编辑器同步);多设备并行;成本与 token 控制。 ## 7. 实现时需新增/改动文件(规划) diff --git a/doc/API.md b/doc/API.md index c143a20..a7c9700 100644 --- a/doc/API.md +++ b/doc/API.md @@ -1193,7 +1193,7 @@ AI 可用设备列表(在线状态 + 是否有任务运行,前端据此把 b 订阅事件流(SSE,EventSource)。事件: - `event: delta` `{text, kind: content|reasoning}` — 流式文本增量 -- `event: step` `{tool, args, image?}` — 工具调用完成(MCP 步骤,image 为缩略截图)。另有两类伪卡片:`tool="🧠 经验记忆"` 表示命中历史经验(args 形如「命中 N 条同类历史经验,已注入参考:<配方摘要>」,N 为实际条数)或本轮已写入经验库 +- `event: step` `{tool, args, image?}` — 工具调用完成(MCP 步骤,image 为缩略截图)。另有三类伪卡片:`tool="🧠 经验记忆"` 表示命中任务级经验(args 形如「命中 N 条同类历史经验,已注入参考:<配方摘要>」)或本轮已写入经验库;`tool="🧠 动作经验"` 表示命中**可复用动作**(「命中 N 个可复用动作,已注入参考:<动作名>」,执行前注入)或本轮已沉淀动作(「已沉淀 N 个可复用动作」,含元素定位、禁坐标) - `event: done` `{answer}` — 完成 - `event: error` `{message}` — 失败(若因 MCP Server 未启动/不可达,message 为明确文案「MCP server(8033) 不可达 …」,不再是 SDK 原始的 `Server returned an error response`) - 空闲时每 15s 发一行 `: keepalive` 注释防超时;`done`/`error` 后关流 @@ -1252,7 +1252,7 @@ AI 可用设备列表(在线状态 + 是否有任务运行,前端据此把 b ### GET /api/agent/experience -经验记忆库列表(自进化,含最近一次巡检结论)。 +经验记忆库列表(自进化,含最近一次巡检结论)。另有一张**动作经验库**表 `agent_action`(命名动作 + 编辑器 schema 步骤 + 元素定位、禁坐标):任务成功后自动从**成功步骤**蒸馏沉淀,执行前按动作名/别名召回并注入;当前无独立查询接口(命中/沉淀在 AI 控制台的 🧠 卡片可见)。 **响应**: ```json diff --git a/doc/ARCHITECTURE.md b/doc/ARCHITECTURE.md index d10b9ba..5915147 100644 --- a/doc/ARCHITECTURE.md +++ b/doc/ARCHITECTURE.md @@ -185,7 +185,8 @@ SQLAlchemy 模型,存于 `data/users.db`: > 表名默认取类名小写(models.py 未写 `__tablename__`)。另有三张非模型表,由原生 SQL 幂等创建、**不走 SCHEMA_MIGRATIONS**: > - `app_meta`(KV):`_migrate_schema()` 内建表,存 `schema_version`、`discovery_*`、agent 配置 `agent_*` 等; -> - `agent_conversation` / `agent_experience` / `experience_audit`:AI 控制台会话 / 经验库 / 经验巡检(`web/agent_api.py` 顶部 `CREATE TABLE IF NOT EXISTS`)。 +> - `agent_conversation` / `agent_experience` / `experience_audit`:AI 控制台会话 / 任务级经验(配方)/ 经验巡检(`web/agent_api.py` 顶部 `CREATE TABLE IF NOT EXISTS`)。 +> - `agent_action`:**动作经验库**(命名动作 = 可复用单元,steps 用编辑器 schema 且带元素定位、禁坐标);由任务成功后从**成功步骤**蒸馏,执行前按名字/别名召回并注入(`web/agent_api.py` `_distill_actions/_find_actions`)。 **数据库初始化**(`init_db`): - 创建所有表 diff --git a/web/agent_api.py b/web/agent_api.py index 29a0fde..5074e55 100644 --- a/web/agent_api.py +++ b/web/agent_api.py @@ -289,6 +289,360 @@ def _save_experience(prompt, recipe, tool_seq): return False +# ---------- 动作经验库(agent_action)---------- +# 与「任务级配方」(agent_experience) 互补:动作 = 有语义名的可复用单元,可含 1~N 步, +# steps 直接用编辑器 schema(open_app/click/input_text…),带**元素定位** +# (selector_type/selector_value),**不含坐标**(分辨率/旋转/改版即失效)。 +# 复用:执行前按 name/别名/App 召回并注入 system prompt,模型可跳过重新探索。 +_ACTION_TABLE = ( + "CREATE TABLE IF NOT EXISTS agent_action (" + "id INTEGER PRIMARY KEY AUTOINCREMENT," + "name TEXT NOT NULL," + "app TEXT DEFAULT ''," + "aliases TEXT DEFAULT '[]'," + "params TEXT DEFAULT '[]'," + "steps TEXT NOT NULL," + "preconditions TEXT DEFAULT ''," + "hits INTEGER DEFAULT 0," + "source_prompt TEXT DEFAULT ''," + "created_at TEXT DEFAULT ''," + "updated_at TEXT DEFAULT '')") + +# 允许沉淀的步骤类型(编辑器 STEP_TYPES 子集;显式排除 click_xy 等坐标类) +_ACTION_STEP_TYPES = { + "open_app", "stop_app", "screen_on", "screen_off", "key_event", + "swipe", "swipe_until", "click", "long_click", "wait_el", + "input_text", "clipboard", "wait", "loop", "group", "if_el"} +_ACTION_REQUIRED = { # 类型 → 必需的 params 键(缺则丢弃该动作) + "open_app": ("package",), "stop_app": ("package",), + "click": ("selector_value",), "long_click": ("selector_value",), + "wait_el": ("selector_value",), "swipe_until": ("selector_value",), + "if_el": ("selector_value", "then"), + "group": ("children",), "loop": ("children",)} +# 可写入动作的关键工具(探索类 de_screenshot/de_ui_tree/de_ocr 不沉淀) +_ACTION_TOOLS = { + "de_open_app", "de_stop_app", "de_tap_text", "de_tap_element", "de_type_text", + "de_set_clipboard", "de_swipe", "de_press_key", "de_wake", "de_sleep"} + + +def _ensure_action_table(): + try: + db.session.execute(db.text(_ACTION_TABLE)) + db.session.commit() + except Exception: + pass + + +def _brief_result(result): + """工具结果的精简摘要(判成败 + 供提炼模型参考)。""" + try: + s = json.dumps(result, ensure_ascii=False) if not isinstance(result, str) else result + except Exception: + s = str(result) + return (s or "")[:160] + + +def _tool_ok(tool, result): + """启发式判断一次工具调用是否成功(只沉淀成功动作,避免把误点当经验)。""" + if result is None: + return False + if isinstance(result, dict): + if result.get("error") or result.get("ok") is False or result.get("success") is False: + return False + if tool == "de_tap_text": + return bool(result.get("matched")) or result.get("method") in ("ui", "ocr") + if result.get("matched") is False: + return False + s = str(result).lower() + return not any(k in s for k in ("not_found", "error", "failed", "occupied")) + + +def _normalize_action_item(a): + """把模型可能的「单动作」形状({action/tool/type, params})归一成标准动作。 + + 实测蒸馏模型常不按提示词的 name/steps 输出,而是照抄输入行给出 + {"action":"de_open_app","params":{...}}——这里做兼容映射,避免全被丢弃。 + """ + if not isinstance(a, dict): + return None + if isinstance(a.get("steps"), list): # 已是标准形状 + return a + tool = str(a.get("action") or a.get("tool") or a.get("type") or "").strip() + p = a.get("params") if isinstance(a.get("params"), dict) else {} + if not tool: + return None + name = str(a.get("name") or "").strip() + t, sp = None, {} + if tool in ("de_open_app", "open_app"): + t, sp = "open_app", {"package": p.get("package", "")} + name = name or f"打开应用 {p.get('package', '')}" + elif tool in ("de_stop_app", "stop_app"): + t, sp = "stop_app", {"package": p.get("package", "")} + name = name or f"关闭应用 {p.get('package', '')}" + elif tool == "de_tap_text": + t, sp = "click", {"selector_type": "text", "selector_value": p.get("text", "")} + name = name or f"点击「{p.get('text', '')}」" + elif tool == "de_tap_element": + by = {"text": "text", "id": "resourceId", "desc": "description", + "text_contains": "text", "desc_contains": "descriptionContains"}.get( + p.get("by"), "text") + t, sp = "click", {"selector_type": by, "selector_value": p.get("value", "")} + name = name or f"点击元素 {p.get('value', '')}" + elif tool == "de_type_text": + t, sp = "input_text", {"mode": "fixed", "fixed_text": p.get("text", "")} + name = name or f"输入「{p.get('text', '')}」" + elif tool == "de_set_clipboard": + t, sp, name = "clipboard", {"text": p.get("text", "")}, name or "写入剪贴板" + elif tool == "de_swipe": + t, sp = "swipe", {"direction": p.get("direction") or "up"} + name = name or "滑动" + elif tool == "de_press_key": + t, sp = "key_event", {"key": p.get("key") or "back"} + name = name or f"按键 {p.get('key', '')}" + elif tool == "de_wake": + t, name = "screen_on", name or "亮屏解锁" + elif tool == "de_sleep": + t, name = "screen_off", name or "息屏" + if not t: + return None + return {"name": name, "app": a.get("app", ""), "aliases": a.get("aliases") or [], + "params": [], "preconditions": a.get("preconditions", ""), + "steps": [{"type": t, "params": sp}]} + + +def _sanitize_actions(actions): + """校验/清洗模型产出的动作:名字非空、步骤白名单+必填、显式禁坐标。""" + out = [] + for a in (actions or []): + a = _normalize_action_item(a) + if not isinstance(a, dict): + continue + name = str(a.get("name") or "").strip()[:40] + steps = a.get("steps") + if not name or not isinstance(steps, list) or not steps: + continue + good = [] + for st in steps: + if not isinstance(st, dict): + continue + t = st.get("type") + if t not in _ACTION_STEP_TYPES: # click_xy 等坐标类在此被剔除 + continue + p = st.get("params") if isinstance(st.get("params"), dict) else {} + if any(not p.get(k) for k in _ACTION_REQUIRED.get(t, ())): + continue + good.append({"type": t, "label": str(st.get("label") or "")[:20], "params": p}) + if not good: + continue + out.append({ + "name": name, + "app": str(a.get("app") or "")[:80], + "aliases": [str(x)[:20] for x in (a.get("aliases") or []) if str(x).strip()][:5], + "params": [str(x)[:20] for x in (a.get("params") or []) if str(x).strip()][:5], + "preconditions": str(a.get("preconditions") or "")[:100], + "steps": good}) + return out + + +def _iter_json_objects(raw): + """从文本里按大括号配对切出顶层 JSON 对象(容忍坏片段,逐条抢救)。""" + depth = 0 + start = None + in_str = esc = False + for i, c in enumerate(raw): + if in_str: + if esc: + esc = False + elif c == "\\": + esc = True + elif c == '"': + in_str = False + continue + if c == '"': + in_str = True + elif c == "{": + if depth == 0: + start = i + depth += 1 + elif c == "}": + depth -= 1 + if depth == 0 and start is not None: + yield raw[start:i + 1] + start = None + + +def _loads_lenient(text): + """尽量解析模型输出的 JSON 数组:容忍 markdown 围栏、尾逗号、中文引号、坏对象。""" + s = (text or "").strip() + s = re.sub(r"^```[a-zA-Z]*\s*|\s*```$", "", s).strip() + m = re.search(r"\[[\s\S]*\]", s) + raw = m.group(0) if m else s + candidates = [raw, + re.sub(r",\s*([\]}])", r"\1", raw), # 去尾逗号 + re.sub(r"[“”]", '"', raw), # 中文引号 → 英文 + re.sub(r"[“”]", '"', re.sub(r",\s*([\]}])", r"\1", raw))] + for cand in candidates: + try: + v = json.loads(cand) + if isinstance(v, list): + return v + except Exception: + continue + # 逐对象抢救(顶层大括号配对),坏的跳过 + out = [] + for obj in _iter_json_objects(raw): + try: + out.append(json.loads(obj)) + except Exception: + continue + return out + + +def _distill_actions(cfg, prompt, trace): + """从**成功**的工具轨迹提炼命名动作(JSON 数组)。失败返回 []。""" + ok_ops = [t for t in (trace or []) + if t.get("tool") in _ACTION_TOOLS and _tool_ok(t.get("tool"), t.get("result"))] + _log.info(f"动作提炼: 轨迹 {len(trace or [])} 步, 成功可沉淀 {len(ok_ops)} 步") + if not ok_ops: + return [] + lines = [f"- {t['tool']} 参数={t['args']} 结果={_brief_result(t['result'])}" for t in ok_ops] + instruction = ( + "以下是一次成功的手机自动化操作的**成功步骤**。请把它们提炼为若干「动作」" + "(每个动作 = 一个有语义名的可复用单元,可含 1~N 步)。只输出 JSON 数组,不要解释。\n" + "字段:name(动作名,如「打开抖音」「搜索关键词」);app(包名,未知则空串);" + "aliases(别名数组);params(参数名数组,如[\"关键词\"]);steps(步骤数组)。\n" + "steps 每步:type + params,type 取值:open_app{package} / click{selector_type," + "selector_value,wait_timeout?} / input_text{mode,fixed_text} / swipe{direction} /" + "wait{min,max} / key_event{key} / group{children}。\n" + "**定位必须用元素定位**:selector_type 取 xpath/text/resourceId/description/" + "descriptionContains,值用上面步骤里出现的真实文字或 id;**禁止坐标**。" + "若某步只能用坐标定位,就不要产出该动作。\n" + "输出示例(**顶层字段必须是 name/steps,禁止用 action/tool 当顶层字段**):\n" + '[{"name":"打开抖音","app":"com.ss.android.ugc.aweme","aliases":["启动抖音"],' + '"params":[],"steps":[{"type":"open_app","params":{"package":"com.ss.android.ugc.aweme"}}]},' + '{"name":"搜索关键词","app":"","aliases":["点搜索"],"params":["关键词"],' + '"steps":[{"type":"click","params":{"selector_type":"text","selector_value":"搜索"}}]}]\n' + f"任务:{prompt[:200]}\n成功步骤:\n" + "\n".join(lines)[:1500]) + try: + import httpx + body = {"model": cfg.get("model") or "deepseek-v4-flash-vision-exp", + "messages": [{"role": "user", "content": instruction}], + "max_tokens": 900} + headers = {"Authorization": f"Bearer {cfg.get('api_key', '')}", + "Content-Type": "application/json"} + r = httpx.post( + f"{(cfg.get('api_base') or 'https://api.deepseek.com').rstrip('/')}/chat/completions", + json=body, headers=headers, timeout=30) + if r.status_code != 200: + _log.warning(f"动作提炼: 模型返回 HTTP {r.status_code}") + return [] + content = (((r.json().get("choices") or [{}])[0].get("message") or {}).get("content") or "") + acts = _sanitize_actions(_loads_lenient(content)) + if not acts: + _log.info(f"动作提炼: 解析后无有效动作(原始输出 {len(content)} 字符)") + return acts + except Exception as e: + _log.warning(f"动作提炼失败: {e}") + return [] + + +def _save_actions(prompt, actions): + """按 (name, app) upsert 保存动作。返回保存条数。""" + if not actions or _flask_app is None: + return 0 + n = 0 + try: + from datetime import datetime + now = datetime.now().strftime("%Y-%m-%d %H:%M") + with _flask_app.app_context(): + _ensure_action_table() + for a in actions: + row = db.session.execute(db.text( + "SELECT id FROM agent_action WHERE name=:n AND app=:a"), + {"n": a["name"], "a": a["app"]}).fetchone() + if row: + db.session.execute(db.text( + "UPDATE agent_action SET aliases=:al, params=:p, steps=:s, " + "preconditions=:pc, updated_at=:t WHERE id=:i"), + {"al": json.dumps(a["aliases"], ensure_ascii=False), + "p": json.dumps(a["params"], ensure_ascii=False), + "s": json.dumps(a["steps"], ensure_ascii=False), + "pc": a["preconditions"], "t": now, "i": row[0]}) + else: + db.session.execute(db.text( + "INSERT INTO agent_action(name, app, aliases, params, steps, " + "preconditions, hits, source_prompt, created_at, updated_at) " + "VALUES(:n,:a,:al,:p,:s,:pc,0,:sp,:t,:t)"), + {"n": a["name"], "a": a["app"], + "al": json.dumps(a["aliases"], ensure_ascii=False), + "p": json.dumps(a["params"], ensure_ascii=False), + "s": json.dumps(a["steps"], ensure_ascii=False), + "pc": a["preconditions"], "sp": prompt[:200], "t": now}) + n += 1 + db.session.commit() + _log.info(f"动作经验已保存 {n} 条") + except Exception as e: + _log.warning(f"动作保存失败: {e}") + return n + + +def _find_actions(prompt, limit=3): + """召回可复用动作:动作名/别名命中 prompt,或与来源提示够相似。返回 (文本, 名字列表)。""" + try: + if _flask_app is None: + return "", [] + with _flask_app.app_context(): + _ensure_action_table() + rows = db.session.execute(db.text( + "SELECT id, name, app, aliases, params, steps, hits FROM agent_action " + "ORDER BY hits DESC, id DESC LIMIT 100")).fetchall() + except Exception: + return "", [] + if not rows: + return "", [] + p_norm = "".join(c for c in (prompt or "").lower() + if c.isalnum() or "一" <= c <= "鿿") + cur = _bigrams(prompt) + scored = [] + for rid, name, app, aliases, params, steps, hits in rows: + try: + keys = [name] + [str(x) for x in (json.loads(aliases) if aliases else [])] + except Exception: + keys = [name] + hit = any(k and k.lower() in p_norm for k in keys if k) + sim = (len(cur & _bigrams(name)) / len(cur)) if cur else 0.0 + if hit or sim >= 0.34: + scored.append((1 if hit else 0, sim, hits or 0, rid, name, app, params, steps)) + if not scored: + return "", [] + scored.sort(key=lambda x: (-x[0], -x[1], -x[2])) + picked = scored[:limit] + try: + with _flask_app.app_context(): + for item in picked: + db.session.execute(db.text("UPDATE agent_action SET hits=hits+1 WHERE id=:i"), + {"i": item[3]}) + db.session.commit() + except Exception: + pass + items, parts = [], [] + for item in picked: + _hit, _sim, _hts, _rid, name, app, params, steps = item + try: + st = json.loads(steps) if steps else [] + except Exception: + st = [] + items.append(name) + brief = " → ".join( + str(s.get("params", {}).get("selector_value") + or s.get("params", {}).get("package") + or s.get("params", {}).get("fixed_text") + or s.get("type")) for s in st[:8]) + parts.append(f"- 「{name}」" + (f"(app={app})" if app else "") + + (f" 参数:{params}" if params else "") + f":{brief}") + return "\n".join(parts), items + + # ================== 经验巡检(AI 质检,删除需人工确认) ================== _audit_state = {"running": False, "last": "", "last_summary": ""} @@ -829,7 +1183,8 @@ def _agent_thread(run_id, prompt, serial, cfg): def on_delta(text, kind): q.put(("delta", {"text": text, "kind": kind})) - tool_seq = [] # 本轮工具序列(经验提炼用) + tool_seq = [] # 本轮工具序列(任务级配方提炼用) + tool_trace = [] # 结构化轨迹:{tool, args, result}(动作经验提炼用,判成败) def on_tool(step): # args 保留对象(json.dumps 序列化)——前端要解析 serial 做画面跟随; @@ -839,11 +1194,16 @@ def _agent_thread(run_id, prompt, serial, cfg): if step.get("image"): rec["image"] = _shrink_image(step["image"]) q.put(("step", rec)) - # 记录精简工具序列 + # 记录精简工具序列 + 结构化轨迹(去 serial、结果截断) try: args = step.get("args") or {} brief = {k: v for k, v in args.items() if k != "serial"} tool_seq.append(f"{step.get('tool')}({str(brief)[:60]})") + res = step.get("result") + # 可沉淀工具保留原始 result(_tool_ok 需按字段判成败);其余只存摘要 + tool_trace.append({"tool": step.get("tool"), "args": brief, + "result": res if step.get("tool") in _ACTION_TOOLS + else _brief_result(res)}) except Exception: pass @@ -885,6 +1245,18 @@ def _agent_thread(run_id, prompt, serial, cfg): "args": f"命中 {len(exp_items)} 条同类历史经验,已注入参考" + (f":{brief}" if brief else ""), "image": None})) + # 动作经验:可复用的命名动作(带元素定位),优先复用可跳过重新探索 + act_ctx, act_items = _find_actions(prompt) + if act_ctx: + _log.info(f"命中可复用动作 {len(act_items)} 个,注入参考") + q.put(("step", {"tool": "🧠 动作经验", + "args": f"命中 {len(act_items)} 个可复用动作,已注入参考" + + (f":{'、'.join(act_items[:3])}" if act_items else ""), + "image": None})) + recall_ctx = exp_ctx + if act_ctx: + recall_ctx += ("\n\n## 可复用动作(优先按其中的元素定位操作;" + "若与当前界面不符,再自行截图确认)\n" + act_ctx) mcp_url = getattr(getattr(agent, "s", None), "mcp_url", "") @@ -914,7 +1286,7 @@ def _agent_thread(run_id, prompt, serial, cfg): on_delta=on_delta, on_tool=on_tool, should_stop=lambda: bool( stop_evt and stop_evt.is_set()), - extra_context=exp_ctx) + extra_context=recall_ctx) except Exception as e: if _mcp_unreachable(e): raise RuntimeError( @@ -961,6 +1333,14 @@ def _agent_thread(run_id, prompt, serial, cfg): "args": "本轮操作已提炼为经验并写入记忆库" "(下次相似任务会自动参考)", "image": None})) + # 动作经验:把本轮**成功**步骤沉淀为命名动作(带元素定位,禁坐标) + acts = _distill_actions(cfg, prompt, tool_trace) + n_act = _save_actions(prompt, acts) + if n_act: + q.put(("step", {"tool": "🧠 动作经验", + "args": f"已沉淀 {n_act} 个可复用动作" + "(含元素定位,下次同类任务可直接复用)", + "image": None})) except Exception as e: _log.warning(f"经验保存异常: {e}") q.put(("done", {"answer": answer}))