From 61808444ad44eb408961c7f37971b08e98ed85bd Mon Sep 17 00:00:00 2001
From: butubb <1422726308@qq.com>
Date: Sat, 10 Oct 2026 21:09:52 +0800
Subject: [PATCH] =?UTF-8?q?feat(monitor):=20=E8=AF=84=E8=AE=BA=E6=8E=A5?=
=?UTF-8?q?=E5=8F=A3=E5=B8=A6=E4=B8=8A=20a=5Fbogus=20=E7=AD=BE=E5=90=8D?=
=?UTF-8?q?=EF=BC=9B=E4=BF=AE=E5=A5=BD=E4=BD=9C=E5=93=81=E5=AF=BC=E5=87=BA?=
=?UTF-8?q?=E7=9A=84=E7=A9=BA=E5=88=97?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
**评论能采了。** 之前 comment/list 一直回 200 + 空 body,被读成「这条没评论」——
而它其实只是被网关挡了。缺的就是 a_bogus 签名,仓库里本来就有
(libs/douyin.js + execjs)。签上之后实测 200 / 9960 字节真评论。
只给评论接口签:作品、详情、博主资料三个不带签名也照常返回,而给它们加签名是
没验证过的改动。签名按需 import —— 那个模块 import 时就把 JS 喂给 execjs,
不该拖进监控层热路径。
**作品导出那几列一直是空的。** 列名写的是裸键 liked_count,而作品行的指标嵌在
metrics / deltas 里,row.get() 永远取到 None —— 导出来的表有「点赞/评论/收藏/
分享」四列,每一格都没有数。原来的测试只断言了「作品ID」,所以没发现。
顺手补上:导出带上 博主备注/昵称、作品备注、发布时间,时间戳格式化成人能读的
形态(原来是一串 13 位毫秒,Excel 里没法看也没法排序)。
---
api/monitor/douyin_api.py | 37 +++++-
api/routers/monitor.py | 65 +++++++--
tests/test_douyin_api.py | 125 ++++++++++++++++++
tests/test_monitor_comments.py | 80 ++++++++++-
.../components/monitor/MonitorDashboard.tsx | 20 ++-
5 files changed, 314 insertions(+), 13 deletions(-)
diff --git a/api/monitor/douyin_api.py b/api/monitor/douyin_api.py
index 3fc98a1..3047b3d 100644
--- a/api/monitor/douyin_api.py
+++ b/api/monitor/douyin_api.py
@@ -41,6 +41,7 @@ import os
import time
from dataclasses import dataclass
from typing import Any, Dict, List, Optional, Tuple
+from urllib.parse import urlencode
import config
import httpx
@@ -281,14 +282,44 @@ def _cookie_value(cookie: str, name: str) -> str:
return ""
+def _sign(params: Dict[str, Any], path: str, user_agent: str) -> Dict[str, Any]:
+ """给一组参数补上 ``a_bogus`` 签名,返回新 dict。
+
+ **按需 import**:那个模块在 import 的那一瞬间就把 ``libs/douyin.js`` 交给 execjs
+ 编译(还要读相对路径),把它拖进监控层的热路径不合适。
+
+ 签名算在**不含 a_bogus 的那串 query 上**,追加到末尾 —— 和爬虫那条路一致,也是
+ 实测能过的形态。
+ """
+ from media_platform.douyin.help import get_a_bogus_from_js
+
+ try:
+ return {
+ **params,
+ "a_bogus": get_a_bogus_from_js(path, urlencode(params), user_agent),
+ }
+ except Exception as exc: # execjs 起不来 / JS 抛错,都算签名失败
+ raise DouyinApiError(f"算 a_bogus 签名失败:{exc}") from exc
+
+
async def _get(
- path: str, params: Dict[str, Any], identity: BrowserIdentity
+ path: str,
+ params: Dict[str, Any],
+ identity: BrowserIdentity,
+ *,
+ signed: bool = False,
) -> Dict[str, Any]:
"""发一个 GET,返回 JSON。
只带调用方给的参数 —— **不要往里加 webid / msToken / browser_version 那一堆**,
那正是爬虫那条路失败的原因。
+
+ ``signed=True`` 时补一个 ``a_bogus``。**只有评论接口需要它**:作品、详情、博主资料
+ 三个不带签名也照常返回,而给它们加签名是没验证过的改动,不做。
"""
+ if signed:
+ params = _sign(params, path, identity.user_agent)
+
url = f"{API_ORIGIN}{path}"
async with httpx.AsyncClient(timeout=REQUEST_TIMEOUT_SECONDS) as client:
response = await client.get(
@@ -487,6 +518,10 @@ async def video_comments(
"aid": 6383,
},
identity,
+ # 这个接口**必须**签名。不签的话网关回 200 + 空 body,会被读成「这条没评论」,
+ # 而它其实只是被挡了 —— 和登录失效长得一模一样。(实测:带上签名 200/9960 字节
+ # 真评论,不带就是空的。)
+ signed=True,
)
records = [
diff --git a/api/routers/monitor.py b/api/routers/monitor.py
index 574e983..7d87558 100644
--- a/api/routers/monitor.py
+++ b/api/routers/monitor.py
@@ -18,7 +18,7 @@
"""HTTP API for scheduled monitoring tasks."""
-from datetime import date, timedelta
+from datetime import date, datetime, timedelta
from typing import Any, Dict, List, Optional
from fastapi import APIRouter, HTTPException, Query, Response
@@ -486,20 +486,29 @@ def _export_columns(kind: str) -> List[tuple[str, str]]:
"""(key, header) pairs per export kind."""
if kind == "notes":
return [
- ("note_id", "作品ID"),
+ # 先放「这是谁」:导出来是拿去比对和汇报的,一行只有作品 ID 没法用。
+ # 备注优先 —— 昵称常常认不出是谁(见 notes 表那一层的说明)。
+ ("creator_alias", "博主备注"),
+ ("creator_name", "博主昵称"),
+ ("note_alias", "作品备注"),
("title", "标题"),
+ ("note_id", "作品ID"),
("note_url", "链接"),
- ("liked_count", "点赞"),
- ("comment_count", "评论"),
- ("collected_count", "收藏"),
- ("share_count", "分享"),
- ("liked_count_delta", "点赞增量"),
- ("comment_count_delta", "评论增量"),
+ ("published_at", "发布时间"),
+ # 指标嵌在 row["metrics"] 里,所以这里必须写成路径 —— 写成裸键名的话这几列
+ # 全空(见 _lookup)。
+ ("metrics.liked_count", "点赞"),
+ ("metrics.comment_count", "评论"),
+ ("metrics.collected_count", "收藏"),
+ ("metrics.share_count", "分享"),
+ ("deltas.liked_count", "点赞增量"),
+ ("deltas.comment_count", "评论增量"),
("first_seen_at", "首次发现"),
("last_seen_at", "最近采集"),
]
if kind == "comments":
return [
+ ("note_creator_name", "博主昵称"),
("note_title", "所属作品"),
("note_id", "作品ID"),
("comment_id", "评论ID"),
@@ -521,6 +530,42 @@ def _export_columns(kind: str) -> List[tuple[str, str]]:
]
+# 表里存的是毫秒时间戳。直接倒进 CSV 就是一串 13 位数字 —— 打开 Excel 的人没法看,
+# 也没法排序。这几个键统一格式化成人能读的形态。
+_TIME_KEYS = {"published_at", "first_seen_at", "last_seen_at", "create_time"}
+
+
+def _fmt_time(value: Any) -> str:
+ """毫秒 → ``YYYY-MM-DD HH:MM``(服务器本地时区)。"""
+ try:
+ return datetime.fromtimestamp(int(value) / 1000).strftime("%Y-%m-%d %H:%M")
+ except (TypeError, ValueError, OSError, OverflowError):
+ return ""
+
+
+def _cell_for(row: Dict[str, Any], key: str) -> Any:
+ """一列的值:时间键格式化成人能读的,其余照原样(None 变空串)。"""
+ value = _lookup(row, key)
+ if key.rsplit(".", 1)[-1] in _TIME_KEYS:
+ return _fmt_time(value)
+ return _cell(value)
+
+
+def _lookup(row: Dict[str, Any], key: str) -> Any:
+ """取一列的值。键可以是 ``metrics.liked_count`` 这种路径。
+
+ 作品行的指标是**嵌在** ``metrics`` / ``deltas`` 里的,而 ``_export_columns`` 里写的
+ 是 ``liked_count`` —— 照顶层键直接 ``row.get()`` 的话,点赞/评论/收藏/分享四列连带
+ 两个增量列**永远是空的**,导出来的表看着有这几列,其实一格都没有。
+ """
+ value: Any = row
+ for part in key.split("."):
+ if not isinstance(value, dict):
+ return None
+ value = value.get(part)
+ return value
+
+
def _cell(value: Any) -> Any:
if value is None:
return ""
@@ -537,7 +582,7 @@ def _to_csv(rows: List[Dict[str, Any]], columns: List[tuple[str, str]]) -> bytes
writer = csv.writer(buffer)
writer.writerow([header for _, header in columns])
for row in rows:
- writer.writerow([_cell(row.get(key)) for key, _ in columns])
+ writer.writerow([_cell_for(row, key) for key, _ in columns])
# utf-8-sig: without the BOM Excel opens Chinese CSV as mojibake, which is
# the single most common complaint about CSV exports here.
@@ -554,7 +599,7 @@ def _to_xlsx(rows: List[Dict[str, Any]], columns: List[tuple[str, str]], sheet:
worksheet.title = {"notes": "作品", "comments": "评论"}.get(sheet, "报表")
worksheet.append([header for _, header in columns])
for row in rows:
- worksheet.append([_cell(row.get(key)) for key, _ in columns])
+ worksheet.append([_cell_for(row, key) for key, _ in columns])
output = io.BytesIO()
workbook.save(output)
diff --git a/tests/test_douyin_api.py b/tests/test_douyin_api.py
index aef3cc0..a67be0c 100644
--- a/tests/test_douyin_api.py
+++ b/tests/test_douyin_api.py
@@ -274,3 +274,128 @@ class TestGet:
assert asyncio.run(douyin_api._get("/x", {}, self._identity())) == {
"user": {"nickname": "x"}
}
+
+
+@pytest.fixture
+def fake_signer(monkeypatch):
+ """把真正的 ``a_bogus`` 签名换成假的。
+
+ 真的那个在 import 的瞬间就要把 ``libs/douyin.js`` 喂给 execjs(还得有 node 和正确的
+ 相对路径),单元测试不该依赖这些。**签名本身是实测过的**:带上它是 200 + 真评论,
+ 不带是 200 + 空 body。这里只负责钉住「有没有带上、传对了没有」。
+ """
+ import sys
+ import types
+
+ module = types.ModuleType("media_platform.douyin.help")
+ calls: list = []
+
+ def get_a_bogus_from_js(url: str, params: str, user_agent: str) -> str:
+ calls.append({"url": url, "params": params, "user_agent": user_agent})
+ return "FAKE-BOGUS"
+
+ module.get_a_bogus_from_js = get_a_bogus_from_js
+ monkeypatch.setitem(sys.modules, "media_platform.douyin.help", module)
+ return calls
+
+
+@pytest.fixture
+def recording_client(monkeypatch):
+ """记下实际发出去的那次请求。"""
+ sent: dict = {}
+
+ class _Response:
+ status_code = 200
+ text = '{"comments": []}'
+
+ def json(self):
+ return {"comments": []}
+
+ class _Client:
+ async def __aenter__(self):
+ return self
+
+ async def __aexit__(self, *exc):
+ return False
+
+ async def get(self, url, **kwargs):
+ sent["url"] = url
+ sent.update(kwargs)
+ return _Response()
+
+ monkeypatch.setattr(httpx, "AsyncClient", lambda **kwargs: _Client())
+ return sent
+
+
+class TestCommentSigning:
+ """评论接口**必须**带 a_bogus。
+
+ 不带的话网关回 200 + 空 body —— 那在下游会变成「这条作品没有评论」,把一次被挡住的
+ 请求伪装成一条正常的空结果。和登录失效长得一模一样,查起来能查半天。
+ """
+
+ def _identity(self):
+ return douyin_api.BrowserIdentity(
+ cookie="sessionid=s", user_agent="UA", client_hints={}
+ )
+
+ def test_the_signature_is_computed_over_the_unsigned_params(
+ self, fake_signer, recording_client
+ ):
+ asyncio.run(
+ douyin_api._get(
+ douyin_api.COMMENT_PATH,
+ {"aweme_id": "123", "count": 20},
+ self._identity(),
+ signed=True,
+ )
+ )
+
+ assert len(fake_signer) == 1
+ # 签名算在**不含 a_bogus** 的那串上 —— 把它自己也算进去是循环的。
+ assert fake_signer[0]["params"] == "aweme_id=123&count=20"
+ assert fake_signer[0]["url"] == douyin_api.COMMENT_PATH
+ assert fake_signer[0]["user_agent"] == "UA"
+
+ def test_the_signature_goes_out_with_the_request(self, fake_signer, recording_client):
+ asyncio.run(
+ douyin_api._get(
+ douyin_api.COMMENT_PATH, {"aweme_id": "123"}, self._identity(), signed=True
+ )
+ )
+
+ assert recording_client["params"]["a_bogus"] == "FAKE-BOGUS"
+ assert recording_client["params"]["aweme_id"] == "123"
+
+ def test_unsigned_calls_never_touch_the_signer(self, fake_signer, recording_client):
+ """作品 / 详情 / 博主资料三个接口不带签名也照常返回。
+
+ 给它们加签名是**没验证过的改动** —— 所以这里钉住「不签」,防止有人图省事把
+ signed=True 改成全局默认。
+ """
+ asyncio.run(
+ douyin_api._get(douyin_api.POSTS_PATH, {"sec_user_id": "x"}, self._identity())
+ )
+
+ assert fake_signer == []
+ assert "a_bogus" not in recording_client["params"]
+
+ def test_a_broken_signer_is_reported_as_such(self, monkeypatch, recording_client):
+ """execjs 起不来时要说出是签名失败,而不是让它变成「没评论」。"""
+ import sys
+ import types
+
+ module = types.ModuleType("media_platform.douyin.help")
+
+ def boom(url, params, user_agent):
+ raise RuntimeError("node 没装")
+
+ module.get_a_bogus_from_js = boom
+ monkeypatch.setitem(sys.modules, "media_platform.douyin.help", module)
+
+ with pytest.raises(douyin_api.DouyinApiError, match="a_bogus"):
+ asyncio.run(
+ douyin_api._get(
+ douyin_api.COMMENT_PATH, {"aweme_id": "1"}, self._identity(), signed=True
+ )
+ )
diff --git a/tests/test_monitor_comments.py b/tests/test_monitor_comments.py
index 6fc83ee..9a689e8 100644
--- a/tests/test_monitor_comments.py
+++ b/tests/test_monitor_comments.py
@@ -20,6 +20,7 @@
import csv
import io
+import re
import httpx
import pytest
@@ -264,7 +265,8 @@ class TestExport:
workbook = load_workbook(io.BytesIO(response.content))
sheet = workbook.active
assert sheet.max_row == 5 # header + four comments
- assert sheet.cell(row=1, column=1).value == "所属作品"
+ assert sheet.cell(row=1, column=1).value == "博主昵称"
+ assert sheet.cell(row=1, column=2).value == "所属作品"
@pytest.mark.asyncio
async def test_report_export(self, client):
@@ -292,3 +294,79 @@ class TestExport:
"/api/monitor/export", params={"kind": "comments", "note_id": "no-such-note"}
)
assert response.status_code == 404
+
+
+class TestNotesExportColumns:
+ """作品导出的列 —— 「有列名」和「列里有数」是两回事。
+
+ 原先这几列写的是裸键名 `liked_count`,而作品行的指标是嵌在 `metrics` 里的,
+ 于是导出来的表有「点赞/评论/收藏/分享」四列,**每一格都是空的**,还没人发现 ——
+ 因为原来的测试只断言了 `作品ID`。
+ """
+
+ @pytest.mark.asyncio
+ async def test_the_metric_columns_actually_contain_numbers(self, client):
+ from api.monitor.models import MonitorNoteMetric
+
+ async with monitor_db.get_session() as session:
+ from sqlalchemy import select
+
+ task_id = (await session.scalar(select(MonitorTask.id))).__int__()
+ session.add(
+ MonitorNoteMetric(
+ task_id=task_id, note_id="note-a", run_id=1,
+ captured_at=1_700_000_000_000,
+ liked_count=123, comment_count=45,
+ collected_count=6, share_count=7,
+ )
+ )
+
+ response = await client.get(
+ "/api/monitor/export", params={"kind": "notes", "format": "csv"}
+ )
+ rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
+ by_id = {row["作品ID"]: row for row in rows}
+
+ assert by_id["note-a"]["点赞"] == "123"
+ assert by_id["note-a"]["评论"] == "45"
+ assert by_id["note-a"]["收藏"] == "6"
+ assert by_id["note-a"]["分享"] == "7"
+
+ @pytest.mark.asyncio
+ async def test_a_missing_metric_is_left_empty_not_zero(self, client):
+ """没采到的指标留空。写 0 的话,导出来的表会声称这条作品零互动。"""
+ response = await client.get(
+ "/api/monitor/export", params={"kind": "notes", "format": "csv"}
+ )
+ rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
+
+ assert rows[0]["点赞"] == ""
+
+ @pytest.mark.asyncio
+ async def test_the_export_says_who_the_creator_is(self, client):
+ """一行只有作品 ID 没法用 —— 导出来是拿去比对和汇报的。
+
+ 备注优先:昵称常常认不出是谁,而备注是人自己起的名字。
+ """
+ response = await client.get(
+ "/api/monitor/export", params={"kind": "notes", "format": "csv"}
+ )
+ rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
+
+ assert {row["博主昵称"] for row in rows} == {"博主甲", "博主乙"}
+
+ @pytest.mark.asyncio
+ async def test_times_are_readable_not_raw_milliseconds(self, client):
+ """毫秒时间戳倒进 CSV 就是 13 位数字,打开 Excel 的人没法看、也没法排序。"""
+ response = await client.get(
+ "/api/monitor/export", params={"kind": "notes", "format": "csv"}
+ )
+ rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
+ by_id = {row["作品ID"]: row for row in rows}
+
+ # 断言**形状**而不是具体时刻:格式化用的是服务器本地时区,写死一个字符串的话
+ # 换个时区的机器上就会红。
+ assert re.fullmatch(r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}", by_id["note-a"]["发布时间"])
+ assert re.fullmatch(r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}", by_id["note-a"]["首次发现"])
+ # 而且不能是原始毫秒。
+ assert by_id["note-a"]["发布时间"] != str(PUBLISHED_A)
diff --git a/webui/src/components/monitor/MonitorDashboard.tsx b/webui/src/components/monitor/MonitorDashboard.tsx
index 9f94cc4..10312b6 100644
--- a/webui/src/components/monitor/MonitorDashboard.tsx
+++ b/webui/src/components/monitor/MonitorDashboard.tsx
@@ -1,10 +1,11 @@
import { useState } from 'react'
-import { Activity, BellRing, FileText, MessageSquare, Plus } from 'lucide-react'
+import { Activity, BellRing, Download, FileText, MessageSquare, Plus } from 'lucide-react'
import { Badge } from '@/components/ui/badge'
import { Button } from '@/components/ui/button'
import { Tabs, TabsContent, TabsList, TabsTrigger } from '@/components/ui/tabs'
import { useMonitorOverview, useMonitorTasks } from '@/hooks/useMonitor'
+import { monitorApi } from '@/lib/api'
import type { MonitorTask } from '@/types/monitor'
import { useCookieStatus, useWebhookStatus } from '@/hooks/useMonitor'
import { useCurrentPlatform } from '@/hooks/usePlatform'
@@ -166,6 +167,23 @@ export function MonitorDashboard() {
>
只看新增
+ {/* 导出的是**这个任务**的全部作品,不是屏幕上这 200 条 —— 屏幕上那份是
+ 为了好看才截断的,导出跟着截断就成了「导出来的比看到的少」。
+ 走 window.open 而不是 blob:鉴权在 Cookie 上,浏览器自己会带上。 */}
+