feat: AI 控制台回答支持 Markdown 渲染 + 推理链可折叠 + token 用量显示
- markdown.js(新增,无 CDN 依赖):轻量 Markdown 渲染(标题/列表含嵌套/表格/ 代码块/引用/链接…),先 esc() 转义再套标记,模型输出的 HTML 只当文本显示 - agent.js:回答改走 Markdown;推理链改为 <details> 可折叠(流式时展开、正文开始 自动收起、手动点过后不再自动改);单条消息 token 脚注 + 顶栏「本会话累计」 - monitor.html:消息结构加 .reasoning/.agent-usage、顶栏 token 徽标、md 相关样式, 引入 markdown.js(base.js 之后、agent.js 之前) - mcp_agent/agent.py:请求带 stream_options.include_usage,按「每次模型调用」累计 usage(末尾 chunk),on_usage 回调吐累计值;网关不认该参数(400/422/点名)时 自动降级重试一次 - web/agent_api.py:SSE 新增 usage 事件、done 带 usage;推理链与用量随会话落库 (_REASONING_KEEP=6000 截断),回灌模型时只取 role/content - 文档:API.md(usage 事件/done/会话消息字段)、ARCHITECTURE §5.4.1、DEVELOPMENT 前端 JS 清单 自测:假模型端点单测 3/3(正常/降级/多轮累加);Edge headless 全链路 27 项全通过 (真实 Flask+SSE+SQLite,含 XSS 转义、刷新后回看);Markdown 渲染器 18 用例全通过
This commit is contained in:
+84
-2
@@ -8,6 +8,8 @@
|
||||
- content/reasoning_content 增量逐 chunk 回调(kind 区分)
|
||||
- tool_calls 分片累积(arguments 按 index 拼接),流结束后统一执行
|
||||
- 截图(de_screenshot)图像转 image_url 追加下一轮,同时 on_tool 回调带缩略
|
||||
- token 用量:请求带 stream_options.include_usage,按「每次模型调用」累计,
|
||||
on_usage 回调吐出累计值(Web 控制台展示)
|
||||
"""
|
||||
import base64
|
||||
import json
|
||||
@@ -22,6 +24,10 @@ _log = logging.getLogger("agent")
|
||||
|
||||
S = AgentSettings()
|
||||
|
||||
|
||||
class _UsageUnsupported(RuntimeError):
|
||||
"""模型/网关不认 stream_options.include_usage(400/422 或报错点名该字段)——降级重试用。"""
|
||||
|
||||
SYSTEM_PROMPT = """你是手机自动化控制助手。你通过工具实时操作 Android 手机。
|
||||
|
||||
工作规范:
|
||||
@@ -56,8 +62,15 @@ class Agent:
|
||||
# 回调(Web 展示用,均可选):
|
||||
# on_delta(text, kind) kind: content | reasoning —— 流式文本增量
|
||||
# on_tool(step) step: {tool, args, result, image} —— 工具调用完成
|
||||
# on_usage(usage) usage: {prompt_tokens, completion_tokens,
|
||||
# total_tokens, calls} —— 累计 token 用量
|
||||
self.on_delta = None
|
||||
self.on_tool = None
|
||||
self.on_usage = None
|
||||
# 本轮累计用量(run_stream 开始时重置)
|
||||
self.usage = {"prompt_tokens": 0, "completion_tokens": 0,
|
||||
"total_tokens": 0, "calls": 0}
|
||||
self._include_usage = True # 模型不认 stream_options 时自动置 False
|
||||
self._mcp = None
|
||||
|
||||
# ---------- MCP 工具桥 ----------
|
||||
@@ -88,7 +101,23 @@ class Agent:
|
||||
|
||||
# ---------- 模型调用(流式) ----------
|
||||
async def _chat_stream(self):
|
||||
"""流式 chat/completions:逐 chunk 产出 JSON(async generator)。"""
|
||||
"""流式 chat/completions:逐 chunk 产出 JSON(async generator)。
|
||||
|
||||
默认要求服务端在末尾 chunk 带 usage(token 统计);个别网关不认
|
||||
`stream_options` 会直接 400,此时自动降级重试一次(不影响主流程)。
|
||||
"""
|
||||
if self._include_usage:
|
||||
try:
|
||||
async for chunk in self._chat_stream_once(True):
|
||||
yield chunk
|
||||
return
|
||||
except _UsageUnsupported as e:
|
||||
_log.warning("模型不支持 stream_options.include_usage,降级重试:%s", e)
|
||||
self._include_usage = False
|
||||
async for chunk in self._chat_stream_once(False):
|
||||
yield chunk
|
||||
|
||||
async def _chat_stream_once(self, with_usage):
|
||||
body = {
|
||||
"model": self.s.model,
|
||||
"messages": self.messages,
|
||||
@@ -96,6 +125,8 @@ class Agent:
|
||||
"max_tokens": 4096,
|
||||
"stream": True,
|
||||
}
|
||||
if with_usage:
|
||||
body["stream_options"] = {"include_usage": True}
|
||||
headers = {"Authorization": f"Bearer {self.s.api_key}",
|
||||
"Content-Type": "application/json"}
|
||||
url = f"{self.s.api_base.rstrip('/')}/chat/completions"
|
||||
@@ -103,6 +134,11 @@ class Agent:
|
||||
async with client.stream("POST", url, json=body, headers=headers) as r:
|
||||
if r.status_code != 200:
|
||||
text = (await r.aread()).decode(errors="replace")
|
||||
# 本次开了 include_usage 却被打回(400/422,或报错里点名这个字段)
|
||||
# → 视作网关不支持,交给上层降级重试(401/余额等真错误照常抛出)
|
||||
if with_usage and (r.status_code in (400, 422)
|
||||
or "stream_options" in text):
|
||||
raise _UsageUnsupported(text[:200])
|
||||
raise RuntimeError(f"模型 API HTTP {r.status_code}: {text[:300]}")
|
||||
async for line in r.aiter_lines():
|
||||
if not line.startswith("data:"):
|
||||
@@ -115,6 +151,37 @@ class Agent:
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
|
||||
# ---------- token 用量 ----------
|
||||
@staticmethod
|
||||
def _read_usage(raw):
|
||||
"""把一次模型调用返回的 usage 规整为 {prompt, completion, total};无效返回 None。"""
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
try:
|
||||
pt = int(raw.get("prompt_tokens") or 0)
|
||||
ct = int(raw.get("completion_tokens") or 0)
|
||||
tt = int(raw.get("total_tokens") or 0) or (pt + ct)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if not (pt or ct or tt):
|
||||
return None
|
||||
return {"prompt_tokens": pt, "completion_tokens": ct, "total_tokens": tt}
|
||||
|
||||
def _accumulate_usage(self, raw):
|
||||
"""把一次模型调用的 usage 累加进本轮总量,并回调 on_usage(累计值)。"""
|
||||
u = self._read_usage(raw)
|
||||
if not u:
|
||||
return
|
||||
self.usage["prompt_tokens"] += u["prompt_tokens"]
|
||||
self.usage["completion_tokens"] += u["completion_tokens"]
|
||||
self.usage["total_tokens"] += u["total_tokens"]
|
||||
self.usage["calls"] += 1
|
||||
if self.on_usage:
|
||||
try:
|
||||
self.on_usage(dict(self.usage))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# ---------- 工具执行 ----------
|
||||
async def _execute_tool(self, name, arguments):
|
||||
"""执行 MCP 工具,返回 (文本结果, image_data_or_None)。"""
|
||||
@@ -161,7 +228,7 @@ class Agent:
|
||||
# ---------- 主循环(流式) ----------
|
||||
async def run_stream(self, prompt: str, serial: str = "",
|
||||
history=None, on_delta=None, on_tool=None,
|
||||
should_stop=None, extra_context=None):
|
||||
should_stop=None, extra_context=None, on_usage=None):
|
||||
"""流式执行一轮指令,返回最终完整文本。
|
||||
|
||||
history:上一轮的 [{"role": "user"|"assistant", "content": 文本}] 列表,
|
||||
@@ -170,9 +237,15 @@ class Agent:
|
||||
on_tool(step):工具调用完成(实时显示 MCP 步骤)
|
||||
should_stop:可调用 fn() -> bool,每轮模型调用前检查(用户中断用)
|
||||
extra_context:附加文本(经验记忆注入,放在 system prompt 末尾)
|
||||
on_usage(usage):每完成一次模型调用回调一次(累计值,见 self.usage)
|
||||
|
||||
本轮累计 token 用量同时留在 self.usage(调用方可直接读)。
|
||||
"""
|
||||
self.on_delta = on_delta
|
||||
self.on_tool = on_tool
|
||||
self.on_usage = on_usage
|
||||
self.usage = {"prompt_tokens": 0, "completion_tokens": 0,
|
||||
"total_tokens": 0, "calls": 0}
|
||||
target = serial or self.s.default_serial
|
||||
sys_txt = SYSTEM_PROMPT
|
||||
if target:
|
||||
@@ -193,9 +266,13 @@ class Agent:
|
||||
tool_acc = {} # index -> {id, name, args}
|
||||
has_tool = False
|
||||
retried = False
|
||||
call_usage = None # 本次模型调用的 usage(末尾 chunk 带)
|
||||
while True:
|
||||
call_usage = None # 重试时丢弃上一次(未完成)的用量
|
||||
try:
|
||||
async for chunk in self._chat_stream():
|
||||
if chunk.get("usage"):
|
||||
call_usage = chunk["usage"]
|
||||
choice = (chunk.get("choices") or [{}])[0]
|
||||
delta = choice.get("delta") or {}
|
||||
text = delta.get("content")
|
||||
@@ -228,6 +305,7 @@ class Agent:
|
||||
continue
|
||||
raise
|
||||
|
||||
self._accumulate_usage(call_usage)
|
||||
full_content = "".join(content_parts)
|
||||
|
||||
if has_tool:
|
||||
@@ -272,13 +350,17 @@ class Agent:
|
||||
"已完成的部分、当前设备状态、未能完成的原因与下一步建议。"
|
||||
"不要调用任何工具。"})
|
||||
parts = []
|
||||
call_usage = None
|
||||
async for chunk in self._chat_stream():
|
||||
if chunk.get("usage"):
|
||||
call_usage = chunk["usage"]
|
||||
delta = (chunk.get("choices") or [{}])[0].get("delta") or {}
|
||||
text = delta.get("content")
|
||||
if text:
|
||||
parts.append(text)
|
||||
if on_delta:
|
||||
on_delta(text, "content")
|
||||
self._accumulate_usage(call_usage)
|
||||
self.tools_schema = saved_tools
|
||||
summary = "".join(parts)
|
||||
return summary or "(已达步骤上限,模型未能生成总结)"
|
||||
|
||||
Reference in New Issue
Block a user