feat(monitor): 评论接口带上 a_bogus 签名;修好作品导出的空列
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s

**评论能采了。** 之前 comment/list 一直回 200 + 空 body,被读成「这条没评论」——
而它其实只是被网关挡了。缺的就是 a_bogus 签名,仓库里本来就有
(libs/douyin.js + execjs)。签上之后实测 200 / 9960 字节真评论。

只给评论接口签:作品、详情、博主资料三个不带签名也照常返回,而给它们加签名是
没验证过的改动。签名按需 import —— 那个模块 import 时就把 JS 喂给 execjs,
不该拖进监控层热路径。

**作品导出那几列一直是空的。** 列名写的是裸键 liked_count,而作品行的指标嵌在
metrics / deltas 里,row.get() 永远取到 None —— 导出来的表有「点赞/评论/收藏/
分享」四列,每一格都没有数。原来的测试只断言了「作品ID」,所以没发现。

顺手补上:导出带上 博主备注/昵称、作品备注、发布时间,时间戳格式化成人能读的
形态(原来是一串 13 位毫秒,Excel 里没法看也没法排序)。
This commit is contained in:
2026-10-10 21:09:52 +08:00
parent e0581682e1
commit 61808444ad
5 changed files with 314 additions and 13 deletions
+125
View File
@@ -274,3 +274,128 @@ class TestGet:
assert asyncio.run(douyin_api._get("/x", {}, self._identity())) == {
"user": {"nickname": "x"}
}
@pytest.fixture
def fake_signer(monkeypatch):
"""把真正的 ``a_bogus`` 签名换成假的。
真的那个在 import 的瞬间就要把 ``libs/douyin.js`` 喂给 execjs(还得有 node 和正确的
相对路径),单元测试不该依赖这些。**签名本身是实测过的**:带上它是 200 + 真评论,
不带是 200 + 空 body。这里只负责钉住「有没有带上、传对了没有」。
"""
import sys
import types
module = types.ModuleType("media_platform.douyin.help")
calls: list = []
def get_a_bogus_from_js(url: str, params: str, user_agent: str) -> str:
calls.append({"url": url, "params": params, "user_agent": user_agent})
return "FAKE-BOGUS"
module.get_a_bogus_from_js = get_a_bogus_from_js
monkeypatch.setitem(sys.modules, "media_platform.douyin.help", module)
return calls
@pytest.fixture
def recording_client(monkeypatch):
"""记下实际发出去的那次请求。"""
sent: dict = {}
class _Response:
status_code = 200
text = '{"comments": []}'
def json(self):
return {"comments": []}
class _Client:
async def __aenter__(self):
return self
async def __aexit__(self, *exc):
return False
async def get(self, url, **kwargs):
sent["url"] = url
sent.update(kwargs)
return _Response()
monkeypatch.setattr(httpx, "AsyncClient", lambda **kwargs: _Client())
return sent
class TestCommentSigning:
"""评论接口**必须**带 a_bogus。
不带的话网关回 200 + 空 body —— 那在下游会变成「这条作品没有评论」,把一次被挡住的
请求伪装成一条正常的空结果。和登录失效长得一模一样,查起来能查半天。
"""
def _identity(self):
return douyin_api.BrowserIdentity(
cookie="sessionid=s", user_agent="UA", client_hints={}
)
def test_the_signature_is_computed_over_the_unsigned_params(
self, fake_signer, recording_client
):
asyncio.run(
douyin_api._get(
douyin_api.COMMENT_PATH,
{"aweme_id": "123", "count": 20},
self._identity(),
signed=True,
)
)
assert len(fake_signer) == 1
# 签名算在**不含 a_bogus** 的那串上 —— 把它自己也算进去是循环的。
assert fake_signer[0]["params"] == "aweme_id=123&count=20"
assert fake_signer[0]["url"] == douyin_api.COMMENT_PATH
assert fake_signer[0]["user_agent"] == "UA"
def test_the_signature_goes_out_with_the_request(self, fake_signer, recording_client):
asyncio.run(
douyin_api._get(
douyin_api.COMMENT_PATH, {"aweme_id": "123"}, self._identity(), signed=True
)
)
assert recording_client["params"]["a_bogus"] == "FAKE-BOGUS"
assert recording_client["params"]["aweme_id"] == "123"
def test_unsigned_calls_never_touch_the_signer(self, fake_signer, recording_client):
"""作品 / 详情 / 博主资料三个接口不带签名也照常返回。
给它们加签名是**没验证过的改动** —— 所以这里钉住「不签」,防止有人图省事把
signed=True 改成全局默认。
"""
asyncio.run(
douyin_api._get(douyin_api.POSTS_PATH, {"sec_user_id": "x"}, self._identity())
)
assert fake_signer == []
assert "a_bogus" not in recording_client["params"]
def test_a_broken_signer_is_reported_as_such(self, monkeypatch, recording_client):
"""execjs 起不来时要说出是签名失败,而不是让它变成「没评论」。"""
import sys
import types
module = types.ModuleType("media_platform.douyin.help")
def boom(url, params, user_agent):
raise RuntimeError("node 没装")
module.get_a_bogus_from_js = boom
monkeypatch.setitem(sys.modules, "media_platform.douyin.help", module)
with pytest.raises(douyin_api.DouyinApiError, match="a_bogus"):
asyncio.run(
douyin_api._get(
douyin_api.COMMENT_PATH, {"aweme_id": "1"}, self._identity(), signed=True
)
)
+79 -1
View File
@@ -20,6 +20,7 @@
import csv
import io
import re
import httpx
import pytest
@@ -264,7 +265,8 @@ class TestExport:
workbook = load_workbook(io.BytesIO(response.content))
sheet = workbook.active
assert sheet.max_row == 5 # header + four comments
assert sheet.cell(row=1, column=1).value == "所属作品"
assert sheet.cell(row=1, column=1).value == "博主昵称"
assert sheet.cell(row=1, column=2).value == "所属作品"
@pytest.mark.asyncio
async def test_report_export(self, client):
@@ -292,3 +294,79 @@ class TestExport:
"/api/monitor/export", params={"kind": "comments", "note_id": "no-such-note"}
)
assert response.status_code == 404
class TestNotesExportColumns:
"""作品导出的列 —— 「有列名」和「列里有数」是两回事。
原先这几列写的是裸键名 `liked_count`,而作品行的指标是嵌在 `metrics` 里的,
于是导出来的表有「点赞/评论/收藏/分享」四列,**每一格都是空的**,还没人发现 ——
因为原来的测试只断言了 `作品ID`。
"""
@pytest.mark.asyncio
async def test_the_metric_columns_actually_contain_numbers(self, client):
from api.monitor.models import MonitorNoteMetric
async with monitor_db.get_session() as session:
from sqlalchemy import select
task_id = (await session.scalar(select(MonitorTask.id))).__int__()
session.add(
MonitorNoteMetric(
task_id=task_id, note_id="note-a", run_id=1,
captured_at=1_700_000_000_000,
liked_count=123, comment_count=45,
collected_count=6, share_count=7,
)
)
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
by_id = {row["作品ID"]: row for row in rows}
assert by_id["note-a"]["点赞"] == "123"
assert by_id["note-a"]["评论"] == "45"
assert by_id["note-a"]["收藏"] == "6"
assert by_id["note-a"]["分享"] == "7"
@pytest.mark.asyncio
async def test_a_missing_metric_is_left_empty_not_zero(self, client):
"""没采到的指标留空。写 0 的话,导出来的表会声称这条作品零互动。"""
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert rows[0]["点赞"] == ""
@pytest.mark.asyncio
async def test_the_export_says_who_the_creator_is(self, client):
"""一行只有作品 ID 没法用 —— 导出来是拿去比对和汇报的。
备注优先:昵称常常认不出是谁,而备注是人自己起的名字。
"""
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert {row["博主昵称"] for row in rows} == {"博主甲", "博主乙"}
@pytest.mark.asyncio
async def test_times_are_readable_not_raw_milliseconds(self, client):
"""毫秒时间戳倒进 CSV 就是 13 位数字,打开 Excel 的人没法看、也没法排序。"""
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
by_id = {row["作品ID"]: row for row in rows}
# 断言**形状**而不是具体时刻:格式化用的是服务器本地时区,写死一个字符串的话
# 换个时区的机器上就会红。
assert re.fullmatch(r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}", by_id["note-a"]["发布时间"])
assert re.fullmatch(r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}", by_id["note-a"]["首次发现"])
# 而且不能是原始毫秒。
assert by_id["note-a"]["发布时间"] != str(PUBLISHED_A)