feat(monitor): 评论接口带上 a_bogus 签名;修好作品导出的空列
**评论能采了。** 之前 comment/list 一直回 200 + 空 body,被读成「这条没评论」—— 而它其实只是被网关挡了。缺的就是 a_bogus 签名,仓库里本来就有 (libs/douyin.js + execjs)。签上之后实测 200 / 9960 字节真评论。 只给评论接口签:作品、详情、博主资料三个不带签名也照常返回,而给它们加签名是 没验证过的改动。签名按需 import —— 那个模块 import 时就把 JS 喂给 execjs, 不该拖进监控层热路径。 **作品导出那几列一直是空的。** 列名写的是裸键 liked_count,而作品行的指标嵌在 metrics / deltas 里,row.get() 永远取到 None —— 导出来的表有「点赞/评论/收藏/ 分享」四列,每一格都没有数。原来的测试只断言了「作品ID」,所以没发现。 顺手补上:导出带上 博主备注/昵称、作品备注、发布时间,时间戳格式化成人能读的 形态(原来是一串 13 位毫秒,Excel 里没法看也没法排序)。
This commit is contained in:
@@ -41,6 +41,7 @@ import os
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
from urllib.parse import urlencode
|
||||
|
||||
import config
|
||||
import httpx
|
||||
@@ -281,14 +282,44 @@ def _cookie_value(cookie: str, name: str) -> str:
|
||||
return ""
|
||||
|
||||
|
||||
def _sign(params: Dict[str, Any], path: str, user_agent: str) -> Dict[str, Any]:
|
||||
"""给一组参数补上 ``a_bogus`` 签名,返回新 dict。
|
||||
|
||||
**按需 import**:那个模块在 import 的那一瞬间就把 ``libs/douyin.js`` 交给 execjs
|
||||
编译(还要读相对路径),把它拖进监控层的热路径不合适。
|
||||
|
||||
签名算在**不含 a_bogus 的那串 query 上**,追加到末尾 —— 和爬虫那条路一致,也是
|
||||
实测能过的形态。
|
||||
"""
|
||||
from media_platform.douyin.help import get_a_bogus_from_js
|
||||
|
||||
try:
|
||||
return {
|
||||
**params,
|
||||
"a_bogus": get_a_bogus_from_js(path, urlencode(params), user_agent),
|
||||
}
|
||||
except Exception as exc: # execjs 起不来 / JS 抛错,都算签名失败
|
||||
raise DouyinApiError(f"算 a_bogus 签名失败:{exc}") from exc
|
||||
|
||||
|
||||
async def _get(
|
||||
path: str, params: Dict[str, Any], identity: BrowserIdentity
|
||||
path: str,
|
||||
params: Dict[str, Any],
|
||||
identity: BrowserIdentity,
|
||||
*,
|
||||
signed: bool = False,
|
||||
) -> Dict[str, Any]:
|
||||
"""发一个 GET,返回 JSON。
|
||||
|
||||
只带调用方给的参数 —— **不要往里加 webid / msToken / browser_version 那一堆**,
|
||||
那正是爬虫那条路失败的原因。
|
||||
|
||||
``signed=True`` 时补一个 ``a_bogus``。**只有评论接口需要它**:作品、详情、博主资料
|
||||
三个不带签名也照常返回,而给它们加签名是没验证过的改动,不做。
|
||||
"""
|
||||
if signed:
|
||||
params = _sign(params, path, identity.user_agent)
|
||||
|
||||
url = f"{API_ORIGIN}{path}"
|
||||
async with httpx.AsyncClient(timeout=REQUEST_TIMEOUT_SECONDS) as client:
|
||||
response = await client.get(
|
||||
@@ -487,6 +518,10 @@ async def video_comments(
|
||||
"aid": 6383,
|
||||
},
|
||||
identity,
|
||||
# 这个接口**必须**签名。不签的话网关回 200 + 空 body,会被读成「这条没评论」,
|
||||
# 而它其实只是被挡了 —— 和登录失效长得一模一样。(实测:带上签名 200/9960 字节
|
||||
# 真评论,不带就是空的。)
|
||||
signed=True,
|
||||
)
|
||||
|
||||
records = [
|
||||
|
||||
+55
-10
@@ -18,7 +18,7 @@
|
||||
|
||||
"""HTTP API for scheduled monitoring tasks."""
|
||||
|
||||
from datetime import date, timedelta
|
||||
from datetime import date, datetime, timedelta
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Query, Response
|
||||
@@ -486,20 +486,29 @@ def _export_columns(kind: str) -> List[tuple[str, str]]:
|
||||
"""(key, header) pairs per export kind."""
|
||||
if kind == "notes":
|
||||
return [
|
||||
("note_id", "作品ID"),
|
||||
# 先放「这是谁」:导出来是拿去比对和汇报的,一行只有作品 ID 没法用。
|
||||
# 备注优先 —— 昵称常常认不出是谁(见 notes 表那一层的说明)。
|
||||
("creator_alias", "博主备注"),
|
||||
("creator_name", "博主昵称"),
|
||||
("note_alias", "作品备注"),
|
||||
("title", "标题"),
|
||||
("note_id", "作品ID"),
|
||||
("note_url", "链接"),
|
||||
("liked_count", "点赞"),
|
||||
("comment_count", "评论"),
|
||||
("collected_count", "收藏"),
|
||||
("share_count", "分享"),
|
||||
("liked_count_delta", "点赞增量"),
|
||||
("comment_count_delta", "评论增量"),
|
||||
("published_at", "发布时间"),
|
||||
# 指标嵌在 row["metrics"] 里,所以这里必须写成路径 —— 写成裸键名的话这几列
|
||||
# 全空(见 _lookup)。
|
||||
("metrics.liked_count", "点赞"),
|
||||
("metrics.comment_count", "评论"),
|
||||
("metrics.collected_count", "收藏"),
|
||||
("metrics.share_count", "分享"),
|
||||
("deltas.liked_count", "点赞增量"),
|
||||
("deltas.comment_count", "评论增量"),
|
||||
("first_seen_at", "首次发现"),
|
||||
("last_seen_at", "最近采集"),
|
||||
]
|
||||
if kind == "comments":
|
||||
return [
|
||||
("note_creator_name", "博主昵称"),
|
||||
("note_title", "所属作品"),
|
||||
("note_id", "作品ID"),
|
||||
("comment_id", "评论ID"),
|
||||
@@ -521,6 +530,42 @@ def _export_columns(kind: str) -> List[tuple[str, str]]:
|
||||
]
|
||||
|
||||
|
||||
# 表里存的是毫秒时间戳。直接倒进 CSV 就是一串 13 位数字 —— 打开 Excel 的人没法看,
|
||||
# 也没法排序。这几个键统一格式化成人能读的形态。
|
||||
_TIME_KEYS = {"published_at", "first_seen_at", "last_seen_at", "create_time"}
|
||||
|
||||
|
||||
def _fmt_time(value: Any) -> str:
|
||||
"""毫秒 → ``YYYY-MM-DD HH:MM``(服务器本地时区)。"""
|
||||
try:
|
||||
return datetime.fromtimestamp(int(value) / 1000).strftime("%Y-%m-%d %H:%M")
|
||||
except (TypeError, ValueError, OSError, OverflowError):
|
||||
return ""
|
||||
|
||||
|
||||
def _cell_for(row: Dict[str, Any], key: str) -> Any:
|
||||
"""一列的值:时间键格式化成人能读的,其余照原样(None 变空串)。"""
|
||||
value = _lookup(row, key)
|
||||
if key.rsplit(".", 1)[-1] in _TIME_KEYS:
|
||||
return _fmt_time(value)
|
||||
return _cell(value)
|
||||
|
||||
|
||||
def _lookup(row: Dict[str, Any], key: str) -> Any:
|
||||
"""取一列的值。键可以是 ``metrics.liked_count`` 这种路径。
|
||||
|
||||
作品行的指标是**嵌在** ``metrics`` / ``deltas`` 里的,而 ``_export_columns`` 里写的
|
||||
是 ``liked_count`` —— 照顶层键直接 ``row.get()`` 的话,点赞/评论/收藏/分享四列连带
|
||||
两个增量列**永远是空的**,导出来的表看着有这几列,其实一格都没有。
|
||||
"""
|
||||
value: Any = row
|
||||
for part in key.split("."):
|
||||
if not isinstance(value, dict):
|
||||
return None
|
||||
value = value.get(part)
|
||||
return value
|
||||
|
||||
|
||||
def _cell(value: Any) -> Any:
|
||||
if value is None:
|
||||
return ""
|
||||
@@ -537,7 +582,7 @@ def _to_csv(rows: List[Dict[str, Any]], columns: List[tuple[str, str]]) -> bytes
|
||||
writer = csv.writer(buffer)
|
||||
writer.writerow([header for _, header in columns])
|
||||
for row in rows:
|
||||
writer.writerow([_cell(row.get(key)) for key, _ in columns])
|
||||
writer.writerow([_cell_for(row, key) for key, _ in columns])
|
||||
|
||||
# utf-8-sig: without the BOM Excel opens Chinese CSV as mojibake, which is
|
||||
# the single most common complaint about CSV exports here.
|
||||
@@ -554,7 +599,7 @@ def _to_xlsx(rows: List[Dict[str, Any]], columns: List[tuple[str, str]], sheet:
|
||||
worksheet.title = {"notes": "作品", "comments": "评论"}.get(sheet, "报表")
|
||||
worksheet.append([header for _, header in columns])
|
||||
for row in rows:
|
||||
worksheet.append([_cell(row.get(key)) for key, _ in columns])
|
||||
worksheet.append([_cell_for(row, key) for key, _ in columns])
|
||||
|
||||
output = io.BytesIO()
|
||||
workbook.save(output)
|
||||
|
||||
@@ -274,3 +274,128 @@ class TestGet:
|
||||
assert asyncio.run(douyin_api._get("/x", {}, self._identity())) == {
|
||||
"user": {"nickname": "x"}
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def fake_signer(monkeypatch):
|
||||
"""把真正的 ``a_bogus`` 签名换成假的。
|
||||
|
||||
真的那个在 import 的瞬间就要把 ``libs/douyin.js`` 喂给 execjs(还得有 node 和正确的
|
||||
相对路径),单元测试不该依赖这些。**签名本身是实测过的**:带上它是 200 + 真评论,
|
||||
不带是 200 + 空 body。这里只负责钉住「有没有带上、传对了没有」。
|
||||
"""
|
||||
import sys
|
||||
import types
|
||||
|
||||
module = types.ModuleType("media_platform.douyin.help")
|
||||
calls: list = []
|
||||
|
||||
def get_a_bogus_from_js(url: str, params: str, user_agent: str) -> str:
|
||||
calls.append({"url": url, "params": params, "user_agent": user_agent})
|
||||
return "FAKE-BOGUS"
|
||||
|
||||
module.get_a_bogus_from_js = get_a_bogus_from_js
|
||||
monkeypatch.setitem(sys.modules, "media_platform.douyin.help", module)
|
||||
return calls
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def recording_client(monkeypatch):
|
||||
"""记下实际发出去的那次请求。"""
|
||||
sent: dict = {}
|
||||
|
||||
class _Response:
|
||||
status_code = 200
|
||||
text = '{"comments": []}'
|
||||
|
||||
def json(self):
|
||||
return {"comments": []}
|
||||
|
||||
class _Client:
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *exc):
|
||||
return False
|
||||
|
||||
async def get(self, url, **kwargs):
|
||||
sent["url"] = url
|
||||
sent.update(kwargs)
|
||||
return _Response()
|
||||
|
||||
monkeypatch.setattr(httpx, "AsyncClient", lambda **kwargs: _Client())
|
||||
return sent
|
||||
|
||||
|
||||
class TestCommentSigning:
|
||||
"""评论接口**必须**带 a_bogus。
|
||||
|
||||
不带的话网关回 200 + 空 body —— 那在下游会变成「这条作品没有评论」,把一次被挡住的
|
||||
请求伪装成一条正常的空结果。和登录失效长得一模一样,查起来能查半天。
|
||||
"""
|
||||
|
||||
def _identity(self):
|
||||
return douyin_api.BrowserIdentity(
|
||||
cookie="sessionid=s", user_agent="UA", client_hints={}
|
||||
)
|
||||
|
||||
def test_the_signature_is_computed_over_the_unsigned_params(
|
||||
self, fake_signer, recording_client
|
||||
):
|
||||
asyncio.run(
|
||||
douyin_api._get(
|
||||
douyin_api.COMMENT_PATH,
|
||||
{"aweme_id": "123", "count": 20},
|
||||
self._identity(),
|
||||
signed=True,
|
||||
)
|
||||
)
|
||||
|
||||
assert len(fake_signer) == 1
|
||||
# 签名算在**不含 a_bogus** 的那串上 —— 把它自己也算进去是循环的。
|
||||
assert fake_signer[0]["params"] == "aweme_id=123&count=20"
|
||||
assert fake_signer[0]["url"] == douyin_api.COMMENT_PATH
|
||||
assert fake_signer[0]["user_agent"] == "UA"
|
||||
|
||||
def test_the_signature_goes_out_with_the_request(self, fake_signer, recording_client):
|
||||
asyncio.run(
|
||||
douyin_api._get(
|
||||
douyin_api.COMMENT_PATH, {"aweme_id": "123"}, self._identity(), signed=True
|
||||
)
|
||||
)
|
||||
|
||||
assert recording_client["params"]["a_bogus"] == "FAKE-BOGUS"
|
||||
assert recording_client["params"]["aweme_id"] == "123"
|
||||
|
||||
def test_unsigned_calls_never_touch_the_signer(self, fake_signer, recording_client):
|
||||
"""作品 / 详情 / 博主资料三个接口不带签名也照常返回。
|
||||
|
||||
给它们加签名是**没验证过的改动** —— 所以这里钉住「不签」,防止有人图省事把
|
||||
signed=True 改成全局默认。
|
||||
"""
|
||||
asyncio.run(
|
||||
douyin_api._get(douyin_api.POSTS_PATH, {"sec_user_id": "x"}, self._identity())
|
||||
)
|
||||
|
||||
assert fake_signer == []
|
||||
assert "a_bogus" not in recording_client["params"]
|
||||
|
||||
def test_a_broken_signer_is_reported_as_such(self, monkeypatch, recording_client):
|
||||
"""execjs 起不来时要说出是签名失败,而不是让它变成「没评论」。"""
|
||||
import sys
|
||||
import types
|
||||
|
||||
module = types.ModuleType("media_platform.douyin.help")
|
||||
|
||||
def boom(url, params, user_agent):
|
||||
raise RuntimeError("node 没装")
|
||||
|
||||
module.get_a_bogus_from_js = boom
|
||||
monkeypatch.setitem(sys.modules, "media_platform.douyin.help", module)
|
||||
|
||||
with pytest.raises(douyin_api.DouyinApiError, match="a_bogus"):
|
||||
asyncio.run(
|
||||
douyin_api._get(
|
||||
douyin_api.COMMENT_PATH, {"aweme_id": "1"}, self._identity(), signed=True
|
||||
)
|
||||
)
|
||||
|
||||
@@ -20,6 +20,7 @@
|
||||
|
||||
import csv
|
||||
import io
|
||||
import re
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
@@ -264,7 +265,8 @@ class TestExport:
|
||||
workbook = load_workbook(io.BytesIO(response.content))
|
||||
sheet = workbook.active
|
||||
assert sheet.max_row == 5 # header + four comments
|
||||
assert sheet.cell(row=1, column=1).value == "所属作品"
|
||||
assert sheet.cell(row=1, column=1).value == "博主昵称"
|
||||
assert sheet.cell(row=1, column=2).value == "所属作品"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_report_export(self, client):
|
||||
@@ -292,3 +294,79 @@ class TestExport:
|
||||
"/api/monitor/export", params={"kind": "comments", "note_id": "no-such-note"}
|
||||
)
|
||||
assert response.status_code == 404
|
||||
|
||||
|
||||
class TestNotesExportColumns:
|
||||
"""作品导出的列 —— 「有列名」和「列里有数」是两回事。
|
||||
|
||||
原先这几列写的是裸键名 `liked_count`,而作品行的指标是嵌在 `metrics` 里的,
|
||||
于是导出来的表有「点赞/评论/收藏/分享」四列,**每一格都是空的**,还没人发现 ——
|
||||
因为原来的测试只断言了 `作品ID`。
|
||||
"""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_the_metric_columns_actually_contain_numbers(self, client):
|
||||
from api.monitor.models import MonitorNoteMetric
|
||||
|
||||
async with monitor_db.get_session() as session:
|
||||
from sqlalchemy import select
|
||||
|
||||
task_id = (await session.scalar(select(MonitorTask.id))).__int__()
|
||||
session.add(
|
||||
MonitorNoteMetric(
|
||||
task_id=task_id, note_id="note-a", run_id=1,
|
||||
captured_at=1_700_000_000_000,
|
||||
liked_count=123, comment_count=45,
|
||||
collected_count=6, share_count=7,
|
||||
)
|
||||
)
|
||||
|
||||
response = await client.get(
|
||||
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
|
||||
)
|
||||
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
|
||||
by_id = {row["作品ID"]: row for row in rows}
|
||||
|
||||
assert by_id["note-a"]["点赞"] == "123"
|
||||
assert by_id["note-a"]["评论"] == "45"
|
||||
assert by_id["note-a"]["收藏"] == "6"
|
||||
assert by_id["note-a"]["分享"] == "7"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_missing_metric_is_left_empty_not_zero(self, client):
|
||||
"""没采到的指标留空。写 0 的话,导出来的表会声称这条作品零互动。"""
|
||||
response = await client.get(
|
||||
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
|
||||
)
|
||||
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
|
||||
|
||||
assert rows[0]["点赞"] == ""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_the_export_says_who_the_creator_is(self, client):
|
||||
"""一行只有作品 ID 没法用 —— 导出来是拿去比对和汇报的。
|
||||
|
||||
备注优先:昵称常常认不出是谁,而备注是人自己起的名字。
|
||||
"""
|
||||
response = await client.get(
|
||||
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
|
||||
)
|
||||
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
|
||||
|
||||
assert {row["博主昵称"] for row in rows} == {"博主甲", "博主乙"}
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_times_are_readable_not_raw_milliseconds(self, client):
|
||||
"""毫秒时间戳倒进 CSV 就是 13 位数字,打开 Excel 的人没法看、也没法排序。"""
|
||||
response = await client.get(
|
||||
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
|
||||
)
|
||||
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
|
||||
by_id = {row["作品ID"]: row for row in rows}
|
||||
|
||||
# 断言**形状**而不是具体时刻:格式化用的是服务器本地时区,写死一个字符串的话
|
||||
# 换个时区的机器上就会红。
|
||||
assert re.fullmatch(r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}", by_id["note-a"]["发布时间"])
|
||||
assert re.fullmatch(r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}", by_id["note-a"]["首次发现"])
|
||||
# 而且不能是原始毫秒。
|
||||
assert by_id["note-a"]["发布时间"] != str(PUBLISHED_A)
|
||||
|
||||
@@ -1,10 +1,11 @@
|
||||
import { useState } from 'react'
|
||||
import { Activity, BellRing, FileText, MessageSquare, Plus } from 'lucide-react'
|
||||
import { Activity, BellRing, Download, FileText, MessageSquare, Plus } from 'lucide-react'
|
||||
|
||||
import { Badge } from '@/components/ui/badge'
|
||||
import { Button } from '@/components/ui/button'
|
||||
import { Tabs, TabsContent, TabsList, TabsTrigger } from '@/components/ui/tabs'
|
||||
import { useMonitorOverview, useMonitorTasks } from '@/hooks/useMonitor'
|
||||
import { monitorApi } from '@/lib/api'
|
||||
import type { MonitorTask } from '@/types/monitor'
|
||||
import { useCookieStatus, useWebhookStatus } from '@/hooks/useMonitor'
|
||||
import { useCurrentPlatform } from '@/hooks/usePlatform'
|
||||
@@ -166,6 +167,23 @@ export function MonitorDashboard() {
|
||||
>
|
||||
只看新增
|
||||
</Button>
|
||||
{/* 导出的是**这个任务**的全部作品,不是屏幕上这 200 条 —— 屏幕上那份是
|
||||
为了好看才截断的,导出跟着截断就成了「导出来的比看到的少」。
|
||||
走 window.open 而不是 blob:鉴权在 Cookie 上,浏览器自己会带上。 */}
|
||||
<Button
|
||||
variant="outline"
|
||||
size="sm"
|
||||
className="ml-auto"
|
||||
onClick={() =>
|
||||
window.open(
|
||||
monitorApi.getExportUrl({ kind: 'notes', taskId: selectedTask.id }),
|
||||
'_blank',
|
||||
)
|
||||
}
|
||||
>
|
||||
<Download className="w-3 h-3 mr-1" />
|
||||
导出 CSV
|
||||
</Button>
|
||||
</div>
|
||||
<NotesTable taskId={selectedTask.id} onlyNew={onlyNew} />
|
||||
</TabsContent>
|
||||
|
||||
Reference in New Issue
Block a user