feat: AI 控制台回答支持 Markdown 渲染 + 推理链可折叠 + token 用量显示
- markdown.js(新增,无 CDN 依赖):轻量 Markdown 渲染(标题/列表含嵌套/表格/ 代码块/引用/链接…),先 esc() 转义再套标记,模型输出的 HTML 只当文本显示 - agent.js:回答改走 Markdown;推理链改为 <details> 可折叠(流式时展开、正文开始 自动收起、手动点过后不再自动改);单条消息 token 脚注 + 顶栏「本会话累计」 - monitor.html:消息结构加 .reasoning/.agent-usage、顶栏 token 徽标、md 相关样式, 引入 markdown.js(base.js 之后、agent.js 之前) - mcp_agent/agent.py:请求带 stream_options.include_usage,按「每次模型调用」累计 usage(末尾 chunk),on_usage 回调吐累计值;网关不认该参数(400/422/点名)时 自动降级重试一次 - web/agent_api.py:SSE 新增 usage 事件、done 带 usage;推理链与用量随会话落库 (_REASONING_KEEP=6000 截断),回灌模型时只取 role/content - 文档:API.md(usage 事件/done/会话消息字段)、ARCHITECTURE §5.4.1、DEVELOPMENT 前端 JS 清单 自测:假模型端点单测 3/3(正常/降级/多轮累加);Edge headless 全链路 27 项全通过 (真实 Flask+SSE+SQLite,含 XSS 转义、刷新后回看);Markdown 渲染器 18 用例全通过
This commit is contained in:
+42
-7
@@ -5,7 +5,9 @@
|
||||
GET /api/agent/stream?run_id= → SSE 事件流(EventSource 订阅):
|
||||
event: delta {text, kind: content|reasoning} 流式文本增量
|
||||
event: step {tool, args, image?} 工具调用完成(MCP 步骤)
|
||||
event: done {answer} 完成
|
||||
event: usage {prompt_tokens, completion_tokens,
|
||||
total_tokens, calls} 本轮累计 token 用量
|
||||
event: done {answer, usage} 完成
|
||||
event: error {message} 失败
|
||||
GET/POST /api/agent/config → 配置读写(key 打码回显)
|
||||
|
||||
@@ -38,9 +40,13 @@ _CFG_KEYS = {"api_base": "agent_api_base",
|
||||
"default_serial": "agent_default_serial",
|
||||
"max_steps": "agent_max_steps"}
|
||||
|
||||
# 推理链落库上限(字符):只留够回看的量,避免会话消息无限膨胀
|
||||
_REASONING_KEEP = 6000
|
||||
|
||||
|
||||
# ---------- 运行状态(单实例 + 事件队列) ----------
|
||||
_run = {"id": None, "state": "idle", "prompt": "", "serial": "",
|
||||
"answer": "", "error": "",
|
||||
"answer": "", "error": "", "usage": {},
|
||||
"history": []} # 多轮对话历史 [{role: user|assistant, content}]
|
||||
_queues = {} # run_id -> queue.Queue(SSE 消费者读取)
|
||||
_stop_events = {} # run_id -> threading.Event(用户中断)
|
||||
@@ -1192,7 +1198,7 @@ def agent_run():
|
||||
_run.update(id=run_id, state="running", prompt=prompt,
|
||||
serial=serial, conv_id=conv_id,
|
||||
started=_dt.now().strftime("%H:%M:%S"),
|
||||
answer="", error="")
|
||||
answer="", error="", usage={})
|
||||
# history 保留(多轮上下文),由会话/「新建会话」管理
|
||||
_queues[run_id] = queue.Queue()
|
||||
_stop_events[run_id] = threading.Event()
|
||||
@@ -1244,6 +1250,7 @@ def agent_run_status():
|
||||
"started": _run.get("started", ""),
|
||||
"answer": (_run.get("answer") or "")[:4000],
|
||||
"error": (_run.get("error") or "")[:400],
|
||||
"usage": _run.get("usage") or {},
|
||||
"history": hist[-16:]})
|
||||
|
||||
|
||||
@@ -1345,9 +1352,22 @@ def _agent_thread(run_id, prompt, serial, cfg):
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
from mcp_agent.agent import Agent
|
||||
|
||||
# 推理链(reasoning_content)累计——单纯流式展示会随页面刷新丢失,
|
||||
# 落库后可在会话里折叠回看(只存文本,不回灌给模型)
|
||||
reason_parts = []
|
||||
reason_len = 0
|
||||
|
||||
def on_delta(text, kind):
|
||||
nonlocal reason_len
|
||||
if kind == "reasoning" and reason_len < _REASONING_KEEP:
|
||||
reason_parts.append(text)
|
||||
reason_len += len(text)
|
||||
q.put(("delta", {"text": text, "kind": kind}))
|
||||
|
||||
def on_usage(usage):
|
||||
"""每次模型调用完成 → 推累计用量(前端实时刷新 token 计数)。"""
|
||||
q.put(("usage", dict(usage)))
|
||||
|
||||
tool_seq = [] # 本轮工具序列(任务级配方提炼用)
|
||||
tool_trace = [] # 结构化轨迹:{tool, args, result}(动作经验提炼用,判成败)
|
||||
|
||||
@@ -1451,7 +1471,8 @@ def _agent_thread(run_id, prompt, serial, cfg):
|
||||
on_delta=on_delta, on_tool=on_tool,
|
||||
should_stop=lambda: bool(
|
||||
stop_evt and stop_evt.is_set()),
|
||||
extra_context=recall_ctx)
|
||||
extra_context=recall_ctx,
|
||||
on_usage=on_usage)
|
||||
except Exception as e:
|
||||
if _mcp_unreachable(e):
|
||||
raise RuntimeError(
|
||||
@@ -1460,13 +1481,22 @@ def _agent_thread(run_id, prompt, serial, cfg):
|
||||
|
||||
# 整体超时保护:卡死时结束,释放单实例
|
||||
answer = asyncio.run(asyncio.wait_for(_execute(), timeout=900))
|
||||
usage = dict(getattr(agent, "usage", None) or {})
|
||||
with _lock:
|
||||
_run["state"] = "done"
|
||||
_run["answer"] = answer
|
||||
# 追加本轮进历史(多轮连续性;上限 12 轮防 token 膨胀)
|
||||
_run["usage"] = usage
|
||||
# 追加本轮进历史(多轮连续性;上限 12 轮防 token 膨胀)。
|
||||
# usage/reasoning 仅用于前端展示与落库,不进模型上下文(读回时只取 role/content)
|
||||
hist = _run.setdefault("history", [])
|
||||
hist.append({"role": "user", "content": prompt[:2000]})
|
||||
hist.append({"role": "assistant", "content": (answer or "")[:4000]})
|
||||
turn = {"role": "assistant", "content": (answer or "")[:4000]}
|
||||
if usage:
|
||||
turn["usage"] = usage
|
||||
reason = "".join(reason_parts).strip()
|
||||
if reason:
|
||||
turn["reasoning"] = reason[:_REASONING_KEEP]
|
||||
hist.append(turn)
|
||||
_run["history"] = hist[-24:]
|
||||
# 会话落库:本轮追加写回(新会话自动用首条消息作标题)。
|
||||
# 后台线程 db 访问需 app context。
|
||||
@@ -1513,7 +1543,7 @@ def _agent_thread(run_id, prompt, serial, cfg):
|
||||
"image": None}))
|
||||
except Exception as e:
|
||||
_log.warning(f"经验保存异常: {e}")
|
||||
q.put(("done", {"answer": answer}))
|
||||
q.put(("done", {"answer": answer, "usage": usage}))
|
||||
except Exception as e:
|
||||
_log.warning(f"Agent 运行异常: {e}")
|
||||
# 诊断:打印消息结构(tool_calls 与 tool 消息配对检查)
|
||||
@@ -1529,6 +1559,11 @@ def _agent_thread(run_id, prompt, serial, cfg):
|
||||
with _lock:
|
||||
_run["state"] = "error"
|
||||
_run["error"] = f"{type(e).__name__}: {str(e)[:200]}"
|
||||
# 失败也保留已花费的 token(前端仍能展示本轮用量;agent 可能未建出来)
|
||||
try:
|
||||
_run["usage"] = dict(agent.usage or {})
|
||||
except Exception:
|
||||
pass
|
||||
q.put(("error", {"message": str(e)[:200]}))
|
||||
finally:
|
||||
q.put(None) # 关闭 SSE
|
||||
|
||||
Reference in New Issue
Block a user