fix: 经验入库质量门槛——①对话续语/质疑/纠错轮次不再存为经验(prompt 过短、继续/还有等续语开头、你确定/还没有/不是吧等质疑词、无任务动词的疑问短句→弃),被用户纠正的失败轮次入库会教坏后续任务;②配方含不存在的 de_* 工具弃存(蒸馏模型编造 de_input 之类);③检索命中 hits 回写 +1,高频有效经验浮前

This commit is contained in:
2026-09-04 15:43:21 +08:00
parent 729f40893e
commit 9c700292c3
+77 -5
View File
@@ -17,6 +17,7 @@ import io
import json
import os
import queue
import re
import sys
import threading
import uuid
@@ -93,15 +94,74 @@ def _bigrams(text):
return {t[i:i + 2] for i in range(len(t) - 1)}
# ---------- 经验质量门槛 ----------
# 自进化经验只应保存「独立操作任务」的成功套路。多轮对话里用户的短句
# (质疑/纠正/催促,如「继续啊」「你确定我是卡1吗」「还没有完成啊」)
# 不是新任务——把执行出错被纠正的轮次存成经验会教坏后续任务。
# 用启发式过滤(零成本,可解释),配方层再校验工具名真实性。
# 任务性动词:命中任一视为有明确操作诉求(疑问/纠错短句一般不含它们)
_TASK_VERBS = ("打开", "搜索", "查看", "找到", "截图", "输入", "点击", "点开",
"发送", "安装", "卸载", "下载", "启动", "停止", "关闭", "退出",
"登录", "切换", "设置", "删除", "清理", "复制", "粘贴", "读取",
"剪贴板", "长按", "滑动", "播放", "发布", "检查", "看看",
"帮我", "给我", "请", "拍张", "查一下")
# 对话续语开头:几乎只出现在承接上一轮(「继续啊」「还有吗」)
_CONTINUE_PREFIXES = ("继续", "还有", "然后呢", "再来", "快点", "刚才",
"接着", "上一步")
# 强质疑/纠错信号(不含「为什么」——「查一下为什么」是正当任务)
_DOUBT_MARKS = ("你确定", "是不是", "不是吧", "不是吗", "怎么都", "怎么还",
"还没有", "没看到", "我说的是", "你听我说", "不对吧", "你又",
"重新来", "错了", "你说得", "你回答")
# 全部真实 MCP 工具(配方里出现不存在的 de_* 说明蒸馏模型在编造,弃存)
_KNOWN_TOOLS = frozenset({
"de_list_devices", "de_screenshot", "de_tap", "de_swipe", "de_ui_tree",
"de_tap_element", "de_read_clipboard", "de_wake", "de_press_key",
"de_open_app", "de_stop_app", "de_foreground_app", "de_type_text",
"de_set_clipboard", "de_sleep", "de_ocr", "de_tap_text", "de_list_apps",
"de_list_tasks"})
def _qualify_experience(prompt, recipe):
"""经验入库前质量门槛,返回 True=值得保存。
1) prompt 太短 / 纯续语开头 / 质疑纠错 → 非独立任务,弃
2) 疑问短句(≤30 字、以 吗/? 结尾)且无任务动词 → 追问/反问,弃
3) 配方含不存在的 de_* 工具(蒸馏模型自由发挥)→ 弃
"""
t = "".join(c for c in (prompt or "") if not c.isspace())
if len(t) < 6:
_log.info("经验弃存:prompt 过短「%s」", t[:20])
return False
if any(t.startswith(p) for p in _CONTINUE_PREFIXES):
_log.info("经验弃存:对话续语开头「%s」", t[:20])
return False
if any(m in t for m in _DOUBT_MARKS):
_log.info("经验弃存:质疑/纠错语气「%s」", t[:20])
return False
if (t.endswith("吗") or t.endswith("?") or t.endswith("?")) \
and len(t) <= 30 and not any(v in t for v in _TASK_VERBS):
_log.info("经验弃存:无操作诉求的追问「%s」", t[:20])
return False
for name in re.findall(r"de_[a-z_]+", recipe or ""):
if name not in _KNOWN_TOOLS:
_log.info("经验弃存:配方含不存在的工具 %s", name)
return False
return True
def _find_experiences(prompt, limit=2, threshold=0.10):
"""按 bigram 重叠检索相似历史经验(prompt 与任务描述的字符相似度)。"""
"""按 bigram 重叠检索相似历史经验(prompt 与任务描述的字符相似度)。
命中的经验 hits+1(回写),让被反复参考的有效经验浮到前面。
"""
try:
if _flask_app is None:
return ""
with _flask_app.app_context():
_ensure_exp_table()
rows = db.session.execute(db.text(
"SELECT task_prompt, recipe, hits FROM agent_experience "
"SELECT id, task_prompt, recipe, hits FROM agent_experience "
"WHERE recipe != '' ORDER BY hits DESC, id DESC LIMIT 50")).fetchall()
except Exception:
return ""
@@ -111,13 +171,24 @@ def _find_experiences(prompt, limit=2, threshold=0.10):
if not cur:
return ""
scored = []
for task_prompt, recipe, hits in rows:
for row_id, task_prompt, recipe, hits in rows:
sim = len(cur & _bigrams(task_prompt)) / len(cur)
if sim >= threshold:
scored.append((sim, hits or 0, recipe))
scored.append((sim, hits or 0, recipe, row_id))
scored.sort(key=lambda x: (-x[0], -x[1]))
if scored:
# hits 回写(尽力而为,失败不影响检索)
try:
with _flask_app.app_context():
for _sim, _hits, _recipe, row_id in scored:
db.session.execute(db.text(
"UPDATE agent_experience SET hits=hits+1 WHERE id=:i"),
{"i": row_id})
db.session.commit()
except Exception:
pass
parts = []
for sim, _hits, recipe in scored[:limit]:
for sim, _hits, recipe, _row_id in scored[:limit]:
parts.append(f"- {recipe[:600]}")
return "\n".join(parts)
@@ -365,6 +436,7 @@ def _agent_thread(run_id, prompt, serial, cfg):
try:
recipe = _distill_experience(cfg, prompt, " -> ".join(tool_seq))
if recipe and "配方" not in recipe[:50]:
if _qualify_experience(prompt, recipe):
if _save_experience(prompt, recipe, " -> ".join(tool_seq)):
_log.info("经验已写入记忆库,随事件流提示")
q.put(("step", {"tool": "🧠 经验记忆",