feat(monitor): 评论接口带上 a_bogus 签名;修好作品导出的空列
Deploy VitePress site to Pages / build (push) Waiting to run
Deploy VitePress site to Pages / Deploy (push) Blocked by required conditions

**评论能采了。** 之前 comment/list 一直回 200 + 空 body,被读成「这条没评论」——
而它其实只是被网关挡了。缺的就是 a_bogus 签名,仓库里本来就有
(libs/douyin.js + execjs)。签上之后实测 200 / 9960 字节真评论。

只给评论接口签:作品、详情、博主资料三个不带签名也照常返回,而给它们加签名是
没验证过的改动。签名按需 import —— 那个模块 import 时就把 JS 喂给 execjs,
不该拖进监控层热路径。

**作品导出那几列一直是空的。** 列名写的是裸键 liked_count,而作品行的指标嵌在
metrics / deltas 里,row.get() 永远取到 None —— 导出来的表有「点赞/评论/收藏/
分享」四列,每一格都没有数。原来的测试只断言了「作品ID」,所以没发现。

顺手补上:导出带上 博主备注/昵称、作品备注、发布时间,时间戳格式化成人能读的
形态(原来是一串 13 位毫秒,Excel 里没法看也没法排序)。
This commit is contained in:
2026-10-10 21:09:52 +08:00
parent e0581682e1
commit 61808444ad
5 changed files with 314 additions and 13 deletions
+36 -1
View File
@@ -41,6 +41,7 @@ import os
import time
from dataclasses import dataclass
from typing import Any, Dict, List, Optional, Tuple
from urllib.parse import urlencode
import config
import httpx
@@ -281,14 +282,44 @@ def _cookie_value(cookie: str, name: str) -> str:
return ""
def _sign(params: Dict[str, Any], path: str, user_agent: str) -> Dict[str, Any]:
"""给一组参数补上 ``a_bogus`` 签名,返回新 dict。
**按需 import**:那个模块在 import 的那一瞬间就把 ``libs/douyin.js`` 交给 execjs
编译(还要读相对路径),把它拖进监控层的热路径不合适。
签名算在**不含 a_bogus 的那串 query 上**,追加到末尾 —— 和爬虫那条路一致,也是
实测能过的形态。
"""
from media_platform.douyin.help import get_a_bogus_from_js
try:
return {
**params,
"a_bogus": get_a_bogus_from_js(path, urlencode(params), user_agent),
}
except Exception as exc: # execjs 起不来 / JS 抛错,都算签名失败
raise DouyinApiError(f"算 a_bogus 签名失败:{exc}") from exc
async def _get(
path: str, params: Dict[str, Any], identity: BrowserIdentity
path: str,
params: Dict[str, Any],
identity: BrowserIdentity,
*,
signed: bool = False,
) -> Dict[str, Any]:
"""发一个 GET,返回 JSON。
只带调用方给的参数 —— **不要往里加 webid / msToken / browser_version 那一堆**,
那正是爬虫那条路失败的原因。
``signed=True`` 时补一个 ``a_bogus``。**只有评论接口需要它**:作品、详情、博主资料
三个不带签名也照常返回,而给它们加签名是没验证过的改动,不做。
"""
if signed:
params = _sign(params, path, identity.user_agent)
url = f"{API_ORIGIN}{path}"
async with httpx.AsyncClient(timeout=REQUEST_TIMEOUT_SECONDS) as client:
response = await client.get(
@@ -487,6 +518,10 @@ async def video_comments(
"aid": 6383,
},
identity,
# 这个接口**必须**签名。不签的话网关回 200 + 空 body,会被读成「这条没评论」,
# 而它其实只是被挡了 —— 和登录失效长得一模一样。(实测:带上签名 200/9960 字节
# 真评论,不带就是空的。)
signed=True,
)
records = [
+55 -10
View File
@@ -18,7 +18,7 @@
"""HTTP API for scheduled monitoring tasks."""
from datetime import date, timedelta
from datetime import date, datetime, timedelta
from typing import Any, Dict, List, Optional
from fastapi import APIRouter, HTTPException, Query, Response
@@ -486,20 +486,29 @@ def _export_columns(kind: str) -> List[tuple[str, str]]:
"""(key, header) pairs per export kind."""
if kind == "notes":
return [
("note_id", "作品ID"),
# 先放「这是谁」:导出来是拿去比对和汇报的,一行只有作品 ID 没法用。
# 备注优先 —— 昵称常常认不出是谁(见 notes 表那一层的说明)。
("creator_alias", "博主备注"),
("creator_name", "博主昵称"),
("note_alias", "作品备注"),
("title", "标题"),
("note_id", "作品ID"),
("note_url", "链接"),
("liked_count", "点赞"),
("comment_count", "评论"),
("collected_count", "收藏"),
("share_count", "分享"),
("liked_count_delta", "点赞增量"),
("comment_count_delta", "评论增量"),
("published_at", "发布时间"),
# 指标嵌在 row["metrics"] 里,所以这里必须写成路径 —— 写成裸键名的话这几列
# 全空(见 _lookup)。
("metrics.liked_count", "点赞"),
("metrics.comment_count", "评论"),
("metrics.collected_count", "收藏"),
("metrics.share_count", "分享"),
("deltas.liked_count", "点赞增量"),
("deltas.comment_count", "评论增量"),
("first_seen_at", "首次发现"),
("last_seen_at", "最近采集"),
]
if kind == "comments":
return [
("note_creator_name", "博主昵称"),
("note_title", "所属作品"),
("note_id", "作品ID"),
("comment_id", "评论ID"),
@@ -521,6 +530,42 @@ def _export_columns(kind: str) -> List[tuple[str, str]]:
]
# 表里存的是毫秒时间戳。直接倒进 CSV 就是一串 13 位数字 —— 打开 Excel 的人没法看,
# 也没法排序。这几个键统一格式化成人能读的形态。
_TIME_KEYS = {"published_at", "first_seen_at", "last_seen_at", "create_time"}
def _fmt_time(value: Any) -> str:
"""毫秒 → ``YYYY-MM-DD HH:MM``(服务器本地时区)。"""
try:
return datetime.fromtimestamp(int(value) / 1000).strftime("%Y-%m-%d %H:%M")
except (TypeError, ValueError, OSError, OverflowError):
return ""
def _cell_for(row: Dict[str, Any], key: str) -> Any:
"""一列的值:时间键格式化成人能读的,其余照原样(None 变空串)。"""
value = _lookup(row, key)
if key.rsplit(".", 1)[-1] in _TIME_KEYS:
return _fmt_time(value)
return _cell(value)
def _lookup(row: Dict[str, Any], key: str) -> Any:
"""取一列的值。键可以是 ``metrics.liked_count`` 这种路径。
作品行的指标是**嵌在** ``metrics`` / ``deltas`` 里的,而 ``_export_columns`` 里写的
是 ``liked_count`` —— 照顶层键直接 ``row.get()`` 的话,点赞/评论/收藏/分享四列连带
两个增量列**永远是空的**,导出来的表看着有这几列,其实一格都没有。
"""
value: Any = row
for part in key.split("."):
if not isinstance(value, dict):
return None
value = value.get(part)
return value
def _cell(value: Any) -> Any:
if value is None:
return ""
@@ -537,7 +582,7 @@ def _to_csv(rows: List[Dict[str, Any]], columns: List[tuple[str, str]]) -> bytes
writer = csv.writer(buffer)
writer.writerow([header for _, header in columns])
for row in rows:
writer.writerow([_cell(row.get(key)) for key, _ in columns])
writer.writerow([_cell_for(row, key) for key, _ in columns])
# utf-8-sig: without the BOM Excel opens Chinese CSV as mojibake, which is
# the single most common complaint about CSV exports here.
@@ -554,7 +599,7 @@ def _to_xlsx(rows: List[Dict[str, Any]], columns: List[tuple[str, str]], sheet:
worksheet.title = {"notes": "作品", "comments": "评论"}.get(sheet, "报表")
worksheet.append([header for _, header in columns])
for row in rows:
worksheet.append([_cell(row.get(key)) for key, _ in columns])
worksheet.append([_cell_for(row, key) for key, _ in columns])
output = io.BytesIO()
workbook.save(output)
+125
View File
@@ -274,3 +274,128 @@ class TestGet:
assert asyncio.run(douyin_api._get("/x", {}, self._identity())) == {
"user": {"nickname": "x"}
}
@pytest.fixture
def fake_signer(monkeypatch):
"""把真正的 ``a_bogus`` 签名换成假的。
真的那个在 import 的瞬间就要把 ``libs/douyin.js`` 喂给 execjs(还得有 node 和正确的
相对路径),单元测试不该依赖这些。**签名本身是实测过的**:带上它是 200 + 真评论,
不带是 200 + 空 body。这里只负责钉住「有没有带上、传对了没有」。
"""
import sys
import types
module = types.ModuleType("media_platform.douyin.help")
calls: list = []
def get_a_bogus_from_js(url: str, params: str, user_agent: str) -> str:
calls.append({"url": url, "params": params, "user_agent": user_agent})
return "FAKE-BOGUS"
module.get_a_bogus_from_js = get_a_bogus_from_js
monkeypatch.setitem(sys.modules, "media_platform.douyin.help", module)
return calls
@pytest.fixture
def recording_client(monkeypatch):
"""记下实际发出去的那次请求。"""
sent: dict = {}
class _Response:
status_code = 200
text = '{"comments": []}'
def json(self):
return {"comments": []}
class _Client:
async def __aenter__(self):
return self
async def __aexit__(self, *exc):
return False
async def get(self, url, **kwargs):
sent["url"] = url
sent.update(kwargs)
return _Response()
monkeypatch.setattr(httpx, "AsyncClient", lambda **kwargs: _Client())
return sent
class TestCommentSigning:
"""评论接口**必须**带 a_bogus。
不带的话网关回 200 + 空 body —— 那在下游会变成「这条作品没有评论」,把一次被挡住的
请求伪装成一条正常的空结果。和登录失效长得一模一样,查起来能查半天。
"""
def _identity(self):
return douyin_api.BrowserIdentity(
cookie="sessionid=s", user_agent="UA", client_hints={}
)
def test_the_signature_is_computed_over_the_unsigned_params(
self, fake_signer, recording_client
):
asyncio.run(
douyin_api._get(
douyin_api.COMMENT_PATH,
{"aweme_id": "123", "count": 20},
self._identity(),
signed=True,
)
)
assert len(fake_signer) == 1
# 签名算在**不含 a_bogus** 的那串上 —— 把它自己也算进去是循环的。
assert fake_signer[0]["params"] == "aweme_id=123&count=20"
assert fake_signer[0]["url"] == douyin_api.COMMENT_PATH
assert fake_signer[0]["user_agent"] == "UA"
def test_the_signature_goes_out_with_the_request(self, fake_signer, recording_client):
asyncio.run(
douyin_api._get(
douyin_api.COMMENT_PATH, {"aweme_id": "123"}, self._identity(), signed=True
)
)
assert recording_client["params"]["a_bogus"] == "FAKE-BOGUS"
assert recording_client["params"]["aweme_id"] == "123"
def test_unsigned_calls_never_touch_the_signer(self, fake_signer, recording_client):
"""作品 / 详情 / 博主资料三个接口不带签名也照常返回。
给它们加签名是**没验证过的改动** —— 所以这里钉住「不签」,防止有人图省事把
signed=True 改成全局默认。
"""
asyncio.run(
douyin_api._get(douyin_api.POSTS_PATH, {"sec_user_id": "x"}, self._identity())
)
assert fake_signer == []
assert "a_bogus" not in recording_client["params"]
def test_a_broken_signer_is_reported_as_such(self, monkeypatch, recording_client):
"""execjs 起不来时要说出是签名失败,而不是让它变成「没评论」。"""
import sys
import types
module = types.ModuleType("media_platform.douyin.help")
def boom(url, params, user_agent):
raise RuntimeError("node 没装")
module.get_a_bogus_from_js = boom
monkeypatch.setitem(sys.modules, "media_platform.douyin.help", module)
with pytest.raises(douyin_api.DouyinApiError, match="a_bogus"):
asyncio.run(
douyin_api._get(
douyin_api.COMMENT_PATH, {"aweme_id": "1"}, self._identity(), signed=True
)
)
+79 -1
View File
@@ -20,6 +20,7 @@
import csv
import io
import re
import httpx
import pytest
@@ -264,7 +265,8 @@ class TestExport:
workbook = load_workbook(io.BytesIO(response.content))
sheet = workbook.active
assert sheet.max_row == 5 # header + four comments
assert sheet.cell(row=1, column=1).value == "所属作品"
assert sheet.cell(row=1, column=1).value == "博主昵称"
assert sheet.cell(row=1, column=2).value == "所属作品"
@pytest.mark.asyncio
async def test_report_export(self, client):
@@ -292,3 +294,79 @@ class TestExport:
"/api/monitor/export", params={"kind": "comments", "note_id": "no-such-note"}
)
assert response.status_code == 404
class TestNotesExportColumns:
"""作品导出的列 —— 「有列名」和「列里有数」是两回事。
原先这几列写的是裸键名 `liked_count`,而作品行的指标是嵌在 `metrics` 里的,
于是导出来的表有「点赞/评论/收藏/分享」四列,**每一格都是空的**,还没人发现 ——
因为原来的测试只断言了 `作品ID`。
"""
@pytest.mark.asyncio
async def test_the_metric_columns_actually_contain_numbers(self, client):
from api.monitor.models import MonitorNoteMetric
async with monitor_db.get_session() as session:
from sqlalchemy import select
task_id = (await session.scalar(select(MonitorTask.id))).__int__()
session.add(
MonitorNoteMetric(
task_id=task_id, note_id="note-a", run_id=1,
captured_at=1_700_000_000_000,
liked_count=123, comment_count=45,
collected_count=6, share_count=7,
)
)
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
by_id = {row["作品ID"]: row for row in rows}
assert by_id["note-a"]["点赞"] == "123"
assert by_id["note-a"]["评论"] == "45"
assert by_id["note-a"]["收藏"] == "6"
assert by_id["note-a"]["分享"] == "7"
@pytest.mark.asyncio
async def test_a_missing_metric_is_left_empty_not_zero(self, client):
"""没采到的指标留空。写 0 的话,导出来的表会声称这条作品零互动。"""
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert rows[0]["点赞"] == ""
@pytest.mark.asyncio
async def test_the_export_says_who_the_creator_is(self, client):
"""一行只有作品 ID 没法用 —— 导出来是拿去比对和汇报的。
备注优先:昵称常常认不出是谁,而备注是人自己起的名字。
"""
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert {row["博主昵称"] for row in rows} == {"博主甲", "博主乙"}
@pytest.mark.asyncio
async def test_times_are_readable_not_raw_milliseconds(self, client):
"""毫秒时间戳倒进 CSV 就是 13 位数字,打开 Excel 的人没法看、也没法排序。"""
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
by_id = {row["作品ID"]: row for row in rows}
# 断言**形状**而不是具体时刻:格式化用的是服务器本地时区,写死一个字符串的话
# 换个时区的机器上就会红。
assert re.fullmatch(r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}", by_id["note-a"]["发布时间"])
assert re.fullmatch(r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}", by_id["note-a"]["首次发现"])
# 而且不能是原始毫秒。
assert by_id["note-a"]["发布时间"] != str(PUBLISHED_A)
@@ -1,10 +1,11 @@
import { useState } from 'react'
import { Activity, BellRing, FileText, MessageSquare, Plus } from 'lucide-react'
import { Activity, BellRing, Download, FileText, MessageSquare, Plus } from 'lucide-react'
import { Badge } from '@/components/ui/badge'
import { Button } from '@/components/ui/button'
import { Tabs, TabsContent, TabsList, TabsTrigger } from '@/components/ui/tabs'
import { useMonitorOverview, useMonitorTasks } from '@/hooks/useMonitor'
import { monitorApi } from '@/lib/api'
import type { MonitorTask } from '@/types/monitor'
import { useCookieStatus, useWebhookStatus } from '@/hooks/useMonitor'
import { useCurrentPlatform } from '@/hooks/usePlatform'
@@ -166,6 +167,23 @@ export function MonitorDashboard() {
>
只看新增
</Button>
{/* 导出的是**这个任务**的全部作品,不是屏幕上这 200 条 —— 屏幕上那份是
为了好看才截断的,导出跟着截断就成了「导出来的比看到的少」。
走 window.open 而不是 blob:鉴权在 Cookie 上,浏览器自己会带上。 */}
<Button
variant="outline"
size="sm"
className="ml-auto"
onClick={() =>
window.open(
monitorApi.getExportUrl({ kind: 'notes', taskId: selectedTask.id }),
'_blank',
)
}
>
<Download className="w-3 h-3 mr-1" />
导出 CSV
</Button>
</div>
<NotesTable taskId={selectedTask.id} onlyNew={onlyNew} />
</TabsContent>