Files
MediaCrawler/tests/test_monitor_comments.py
T
butubb 61808444ad
Deploy VitePress site to Pages / build (push) Waiting to run
Deploy VitePress site to Pages / Deploy (push) Blocked by required conditions
feat(monitor): 评论接口带上 a_bogus 签名;修好作品导出的空列
**评论能采了。** 之前 comment/list 一直回 200 + 空 body,被读成「这条没评论」——
而它其实只是被网关挡了。缺的就是 a_bogus 签名,仓库里本来就有
(libs/douyin.js + execjs)。签上之后实测 200 / 9960 字节真评论。

只给评论接口签:作品、详情、博主资料三个不带签名也照常返回,而给它们加签名是
没验证过的改动。签名按需 import —— 那个模块 import 时就把 JS 喂给 execjs,
不该拖进监控层热路径。

**作品导出那几列一直是空的。** 列名写的是裸键 liked_count,而作品行的指标嵌在
metrics / deltas 里,row.get() 永远取到 None —— 导出来的表有「点赞/评论/收藏/
分享」四列,每一格都没有数。原来的测试只断言了「作品ID」,所以没发现。

顺手补上:导出带上 博主备注/昵称、作品备注、发布时间,时间戳格式化成人能读的
形态(原来是一串 13 位毫秒,Excel 里没法看也没法排序)。
2026-10-10 21:09:52 +08:00

373 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/tests/test_monitor_comments.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""Comment note-association, grouping, and the export endpoint."""
import csv
import io
import re
import httpx
import pytest
import pytest_asyncio
from api.main import app
from api.monitor import db as monitor_db
from api.monitor.models import (
MODE_CREATOR,
MonitorComment,
MonitorNote,
MonitorTask,
)
TASK_NAME = "评论归属测试"
# 作品的发布时间。和 first_seen_at(我们第一次看到它)刻意取不同的值 —— 两者混成
# 一个概念是最容易犯的错。
PUBLISHED_A = 1_699_000_000_000
PUBLISHED_B = 1_699_100_000_000
async def _seed():
"""Two works; three comments on the first, one on the second."""
async with monitor_db.get_session() as session:
task = MonitorTask(
name=TASK_NAME, platform="xhs", mode=MODE_CREATOR, enabled=True,
interval_minutes=60, max_notes_count=20, enable_comments=True,
max_comments_count=50, run_timeout_seconds=3600,
notify_enabled=False, created_at=0, updated_at=0,
)
session.add(task)
await session.flush()
# 两个作品**属于不同的博主** —— 评论流最外层按创作者分组,同一个人就没得测了。
for note_id, title, creator_hash, creator_name, published_at in (
("note-a", "作品甲", "hash-a", "博主甲", PUBLISHED_A),
("note-b", "作品乙", "hash-b", "博主乙", PUBLISHED_B),
):
session.add(
MonitorNote(
task_id=task.id, note_id=note_id, title=title,
note_url=f"https://www.xiaohongshu.com/explore/{note_id}",
cover=f"https://img/{note_id}.jpg", creator_hash=creator_hash,
creator_name=creator_name,
source_kind="video", published_at=published_at,
first_seen_run_id=1, first_seen_at=1_700_000_000_000,
last_seen_run_id=1, last_seen_at=1_700_000_000_000,
)
)
# note-a has three comments, note-b has one.
plan = [
("c1", "note-a", 1_700_000_001_000),
("c2", "note-a", 1_700_000_002_000),
("c3", "note-a", 1_700_000_003_000),
("c4", "note-b", 1_700_000_004_000),
]
for comment_id, note_id, seen_at in plan:
session.add(
MonitorComment(
task_id=task.id, note_id=note_id, comment_id=comment_id,
content=f"内容-{comment_id}", nickname="u***r", creator_hash="h",
create_time=seen_at, like_count=1, sub_comment_count=0,
parent_comment_id="", first_seen_run_id=1, first_seen_at=seen_at,
)
)
return task.id
@pytest_asyncio.fixture
async def client(tmp_path):
monitor_db.set_sqlite_path(tmp_path / "monitor.db")
await monitor_db.init_db()
await _seed()
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url="http://test") as http_client:
yield http_client
await monitor_db.dispose_engine()
class TestCommentsCarryTheirNote:
@pytest.mark.asyncio
async def test_each_comment_names_its_work(self, client):
"""A bare note_id is unreadable -- the title is the whole point."""
response = await client.get("/api/monitor/comments")
assert response.status_code == 200
comments = response.json()["comments"]
assert len(comments) == 4
by_id = {c["comment_id"]: c for c in comments}
assert by_id["c1"]["note_title"] == "作品甲"
assert by_id["c1"]["note_url"].endswith("note-a")
assert by_id["c1"]["note_cover"].endswith("note-a.jpg")
assert by_id["c4"]["note_title"] == "作品乙"
@pytest.mark.asyncio
async def test_note_id_filters_the_stream(self, client):
response = await client.get("/api/monitor/comments", params={"note_id": "note-a"})
comments = response.json()["comments"]
assert {c["comment_id"] for c in comments} == {"c1", "c2", "c3"}
class TestGroupByNote:
@pytest.mark.asyncio
async def test_groups_bucket_by_work(self, client):
response = await client.get("/api/monitor/comments", params={"group_by": "note"})
body = response.json()
assert "groups" in body
assert body["total"] == 4
groups = {g["note_id"]: g for g in body["groups"]}
assert set(groups) == {"note-a", "note-b"}
assert len(groups["note-a"]["comments"]) == 3
assert len(groups["note-b"]["comments"]) == 1
assert groups["note-a"]["note_title"] == "作品甲"
@pytest.mark.asyncio
async def test_newest_group_comes_first(self, client):
"""The UI expands the first group by default, so it must be the newest."""
response = await client.get("/api/monitor/comments", params={"group_by": "note"})
groups = response.json()["groups"]
# note-b's only comment is the most recent overall.
assert groups[0]["note_id"] == "note-b"
@pytest.mark.asyncio
async def test_each_bucket_carries_its_creator(self, client):
"""桶上必须带作品的创作者 —— 评论流最外层就是按它分组的。
少了这两个字段,前端拿到的 creator_hash / creator_name 都是 undefined,
于是所有博主塌成同一个分组、标签回退成「未知博主」:一个人都分不出来。
"""
groups = {
group["note_id"]: group
for group in (
await client.get("/api/monitor/comments", params={"group_by": "note"})
).json()["groups"]
}
assert groups["note-a"]["creator_hash"] == "hash-a"
assert groups["note-a"]["creator_name"] == "博主甲"
assert groups["note-b"]["creator_hash"] == "hash-b"
assert groups["note-b"]["creator_name"] == "博主乙"
# 两个作品的创作者必须真的不同,否则界面上照样分不出来。
assert groups["note-a"]["creator_hash"] != groups["note-b"]["creator_hash"]
@pytest.mark.asyncio
async def test_each_bucket_carries_the_publish_date(self, client):
"""作品那一层要带发布日期:同名作品不少,日期能帮着认。
注意它和 first_seen_at 是两个概念 —— 前者是作者发布的那天,后者是我们第一次
看到它的那天。把老作品加进监控时两者能差好几个月。
"""
groups = {
group["note_id"]: group
for group in (
await client.get("/api/monitor/comments", params={"group_by": "note"})
).json()["groups"]
}
assert groups["note-a"]["published_at"] == PUBLISHED_A
assert groups["note-b"]["published_at"] == PUBLISHED_B
assert groups["note-a"]["published_at"] != groups["note-b"]["published_at"]
@pytest.mark.asyncio
async def test_the_notes_endpoint_exposes_the_publish_date(self, client):
"""作品列表也要带上它 —— 作品栏就是靠这个显示「发布日期」列的。"""
notes = {
note["note_id"]: note
for note in (await client.get("/api/monitor/notes")).json()["notes"]
}
assert notes["note-a"]["published_at"] == PUBLISHED_A
assert notes["note-b"]["published_at"] == PUBLISHED_B
# 和「首次发现」不是同一个值 —— 两者混了的话这个断言会抓到。
assert notes["note-a"]["first_seen_at"] != notes["note-a"]["published_at"]
@pytest.mark.asyncio
async def test_flat_shape_is_unchanged_without_the_flag(self, client):
body = (await client.get("/api/monitor/comments")).json()
assert "comments" in body and "groups" not in body
class TestCommentNoteFilterOptions:
@pytest.mark.asyncio
async def test_options_carry_counts_and_titles(self, client):
response = await client.get("/api/monitor/comment-notes")
assert response.status_code == 200
notes = {n["note_id"]: n for n in response.json()["notes"]}
assert notes["note-a"]["comment_count"] == 3
assert notes["note-b"]["comment_count"] == 1
assert notes["note-a"]["note_title"] == "作品甲"
@pytest.mark.asyncio
async def test_scoped_to_a_task(self, client):
tasks = (await client.get("/api/monitor/tasks")).json()["tasks"]
task_id = tasks[0]["id"]
scoped = await client.get("/api/monitor/comment-notes", params={"task_id": task_id})
assert len(scoped.json()["notes"]) == 2
# A task with no comments yields an empty list, not an error.
other = await client.get("/api/monitor/comment-notes", params={"task_id": 9999})
assert other.json()["notes"] == []
class TestExport:
@pytest.mark.asyncio
async def test_csv_has_a_bom_so_excel_does_not_mangle_chinese(self, client):
response = await client.get("/api/monitor/export", params={"kind": "comments"})
assert response.status_code == 200
assert response.content.startswith(b"\xef\xbb\xbf")
assert "attachment" in response.headers["content-disposition"]
text = response.content.decode("utf-8-sig")
rows = list(csv.DictReader(io.StringIO(text)))
assert len(rows) == 4
assert rows[0]["所属作品"] in ("作品甲", "作品乙")
@pytest.mark.asyncio
async def test_notes_export(self, client):
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert {r["作品ID"] for r in rows} == {"note-a", "note-b"}
@pytest.mark.asyncio
async def test_xlsx_is_a_readable_workbook(self, client):
from openpyxl import load_workbook
response = await client.get(
"/api/monitor/export", params={"kind": "comments", "format": "xlsx"}
)
assert response.status_code == 200
workbook = load_workbook(io.BytesIO(response.content))
sheet = workbook.active
assert sheet.max_row == 5 # header + four comments
assert sheet.cell(row=1, column=1).value == "博主昵称"
assert sheet.cell(row=1, column=2).value == "所属作品"
@pytest.mark.asyncio
async def test_report_export(self, client):
response = await client.get(
"/api/monitor/export",
params={"kind": "report", "days": 3},
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert len(rows) == 3
assert "日期" in rows[0]
@pytest.mark.asyncio
async def test_unknown_kind_and_format_are_rejected(self, client):
assert (
await client.get("/api/monitor/export", params={"kind": "nope"})
).status_code == 400
assert (
await client.get("/api/monitor/export", params={"kind": "notes", "format": "pdf"})
).status_code == 400
@pytest.mark.asyncio
async def test_empty_selection_is_a_404_not_an_empty_file(self, client):
"""An empty download looks like a bug; say so instead."""
response = await client.get(
"/api/monitor/export", params={"kind": "comments", "note_id": "no-such-note"}
)
assert response.status_code == 404
class TestNotesExportColumns:
"""作品导出的列 —— 「有列名」和「列里有数」是两回事。
原先这几列写的是裸键名 `liked_count`,而作品行的指标是嵌在 `metrics` 里的,
于是导出来的表有「点赞/评论/收藏/分享」四列,**每一格都是空的**,还没人发现 ——
因为原来的测试只断言了 `作品ID`。
"""
@pytest.mark.asyncio
async def test_the_metric_columns_actually_contain_numbers(self, client):
from api.monitor.models import MonitorNoteMetric
async with monitor_db.get_session() as session:
from sqlalchemy import select
task_id = (await session.scalar(select(MonitorTask.id))).__int__()
session.add(
MonitorNoteMetric(
task_id=task_id, note_id="note-a", run_id=1,
captured_at=1_700_000_000_000,
liked_count=123, comment_count=45,
collected_count=6, share_count=7,
)
)
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
by_id = {row["作品ID"]: row for row in rows}
assert by_id["note-a"]["点赞"] == "123"
assert by_id["note-a"]["评论"] == "45"
assert by_id["note-a"]["收藏"] == "6"
assert by_id["note-a"]["分享"] == "7"
@pytest.mark.asyncio
async def test_a_missing_metric_is_left_empty_not_zero(self, client):
"""没采到的指标留空。写 0 的话,导出来的表会声称这条作品零互动。"""
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert rows[0]["点赞"] == ""
@pytest.mark.asyncio
async def test_the_export_says_who_the_creator_is(self, client):
"""一行只有作品 ID 没法用 —— 导出来是拿去比对和汇报的。
备注优先:昵称常常认不出是谁,而备注是人自己起的名字。
"""
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert {row["博主昵称"] for row in rows} == {"博主甲", "博主乙"}
@pytest.mark.asyncio
async def test_times_are_readable_not_raw_milliseconds(self, client):
"""毫秒时间戳倒进 CSV 就是 13 位数字,打开 Excel 的人没法看、也没法排序。"""
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
by_id = {row["作品ID"]: row for row in rows}
# 断言**形状**而不是具体时刻:格式化用的是服务器本地时区,写死一个字符串的话
# 换个时区的机器上就会红。
assert re.fullmatch(r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}", by_id["note-a"]["发布时间"])
assert re.fullmatch(r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}", by_id["note-a"]["首次发现"])
# 而且不能是原始毫秒。
assert by_id["note-a"]["发布时间"] != str(PUBLISHED_A)