Files
MediaCrawler/tests/test_monitor_comments.py
T
butubb 67837b407e
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
fix(comments): 评论栏所有博主都显示成「未知博主」
现象:小红书的和抖音的评论栏,最外层分组全是「未知博主」,分不出谁是谁。

根因是分组接口漏了两个字段。api/routers/monitor.py 里按作品构造桶时只放了
note_id/note_title/note_cover/note_url/comments,而前端 groupByCreator 是用
bucket.creator_hash / bucket.creator_name 分组的 —— 两个都是 undefined,于是所有
博主塌成同一个 key,标签取空串回退成「未知博主」。

数据一直都在:每条评论上都带着 note_creator_hash / note_creator_name
(service.py:540-541),只是没往桶上搬。TS 的 CommentBucket 里也声明了这两个字段,
所以是后端没兑现自己的契约,不是前端写错。

修:构造桶时把作品的创作者一并放上去(同一个桶里的评论必然同属一个作品,取哪条都一样)。

测试:种子数据改成「两个作品属于不同博主」(原来是同一个 hash,测不出这个 bug),
新增一条断言每个桶带上自己那个博主、且两个博主的 hash 确实不同。
2026-10-10 15:41:42 +08:00

259 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/tests/test_monitor_comments.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""Comment note-association, grouping, and the export endpoint."""
import csv
import io
import httpx
import pytest
import pytest_asyncio
from api.main import app
from api.monitor import db as monitor_db
from api.monitor.models import (
MODE_CREATOR,
MonitorComment,
MonitorNote,
MonitorTask,
)
TASK_NAME = "评论归属测试"
async def _seed():
"""Two works; three comments on the first, one on the second."""
async with monitor_db.get_session() as session:
task = MonitorTask(
name=TASK_NAME, platform="xhs", mode=MODE_CREATOR, enabled=True,
interval_minutes=60, max_notes_count=20, enable_comments=True,
max_comments_count=50, run_timeout_seconds=3600,
notify_enabled=False, created_at=0, updated_at=0,
)
session.add(task)
await session.flush()
# 两个作品**属于不同的博主** —— 评论流最外层按创作者分组,同一个人就没得测了。
for note_id, title, creator_hash, creator_name in (
("note-a", "作品甲", "hash-a", "博主甲"),
("note-b", "作品乙", "hash-b", "博主乙"),
):
session.add(
MonitorNote(
task_id=task.id, note_id=note_id, title=title,
note_url=f"https://www.xiaohongshu.com/explore/{note_id}",
cover=f"https://img/{note_id}.jpg", creator_hash=creator_hash,
creator_name=creator_name,
source_kind="video", published_at=None,
first_seen_run_id=1, first_seen_at=1_700_000_000_000,
last_seen_run_id=1, last_seen_at=1_700_000_000_000,
)
)
# note-a has three comments, note-b has one.
plan = [
("c1", "note-a", 1_700_000_001_000),
("c2", "note-a", 1_700_000_002_000),
("c3", "note-a", 1_700_000_003_000),
("c4", "note-b", 1_700_000_004_000),
]
for comment_id, note_id, seen_at in plan:
session.add(
MonitorComment(
task_id=task.id, note_id=note_id, comment_id=comment_id,
content=f"内容-{comment_id}", nickname="u***r", creator_hash="h",
create_time=seen_at, like_count=1, sub_comment_count=0,
parent_comment_id="", first_seen_run_id=1, first_seen_at=seen_at,
)
)
return task.id
@pytest_asyncio.fixture
async def client(tmp_path):
monitor_db.set_sqlite_path(tmp_path / "monitor.db")
await monitor_db.init_db()
await _seed()
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url="http://test") as http_client:
yield http_client
await monitor_db.dispose_engine()
class TestCommentsCarryTheirNote:
@pytest.mark.asyncio
async def test_each_comment_names_its_work(self, client):
"""A bare note_id is unreadable -- the title is the whole point."""
response = await client.get("/api/monitor/comments")
assert response.status_code == 200
comments = response.json()["comments"]
assert len(comments) == 4
by_id = {c["comment_id"]: c for c in comments}
assert by_id["c1"]["note_title"] == "作品甲"
assert by_id["c1"]["note_url"].endswith("note-a")
assert by_id["c1"]["note_cover"].endswith("note-a.jpg")
assert by_id["c4"]["note_title"] == "作品乙"
@pytest.mark.asyncio
async def test_note_id_filters_the_stream(self, client):
response = await client.get("/api/monitor/comments", params={"note_id": "note-a"})
comments = response.json()["comments"]
assert {c["comment_id"] for c in comments} == {"c1", "c2", "c3"}
class TestGroupByNote:
@pytest.mark.asyncio
async def test_groups_bucket_by_work(self, client):
response = await client.get("/api/monitor/comments", params={"group_by": "note"})
body = response.json()
assert "groups" in body
assert body["total"] == 4
groups = {g["note_id"]: g for g in body["groups"]}
assert set(groups) == {"note-a", "note-b"}
assert len(groups["note-a"]["comments"]) == 3
assert len(groups["note-b"]["comments"]) == 1
assert groups["note-a"]["note_title"] == "作品甲"
@pytest.mark.asyncio
async def test_newest_group_comes_first(self, client):
"""The UI expands the first group by default, so it must be the newest."""
response = await client.get("/api/monitor/comments", params={"group_by": "note"})
groups = response.json()["groups"]
# note-b's only comment is the most recent overall.
assert groups[0]["note_id"] == "note-b"
@pytest.mark.asyncio
async def test_each_bucket_carries_its_creator(self, client):
"""桶上必须带作品的创作者 —— 评论流最外层就是按它分组的。
少了这两个字段,前端拿到的 creator_hash / creator_name 都是 undefined,
于是所有博主塌成同一个分组、标签回退成「未知博主」:一个人都分不出来。
"""
groups = {
group["note_id"]: group
for group in (
await client.get("/api/monitor/comments", params={"group_by": "note"})
).json()["groups"]
}
assert groups["note-a"]["creator_hash"] == "hash-a"
assert groups["note-a"]["creator_name"] == "博主甲"
assert groups["note-b"]["creator_hash"] == "hash-b"
assert groups["note-b"]["creator_name"] == "博主乙"
# 两个作品的创作者必须真的不同,否则界面上照样分不出来。
assert groups["note-a"]["creator_hash"] != groups["note-b"]["creator_hash"]
@pytest.mark.asyncio
async def test_flat_shape_is_unchanged_without_the_flag(self, client):
body = (await client.get("/api/monitor/comments")).json()
assert "comments" in body and "groups" not in body
class TestCommentNoteFilterOptions:
@pytest.mark.asyncio
async def test_options_carry_counts_and_titles(self, client):
response = await client.get("/api/monitor/comment-notes")
assert response.status_code == 200
notes = {n["note_id"]: n for n in response.json()["notes"]}
assert notes["note-a"]["comment_count"] == 3
assert notes["note-b"]["comment_count"] == 1
assert notes["note-a"]["note_title"] == "作品甲"
@pytest.mark.asyncio
async def test_scoped_to_a_task(self, client):
tasks = (await client.get("/api/monitor/tasks")).json()["tasks"]
task_id = tasks[0]["id"]
scoped = await client.get("/api/monitor/comment-notes", params={"task_id": task_id})
assert len(scoped.json()["notes"]) == 2
# A task with no comments yields an empty list, not an error.
other = await client.get("/api/monitor/comment-notes", params={"task_id": 9999})
assert other.json()["notes"] == []
class TestExport:
@pytest.mark.asyncio
async def test_csv_has_a_bom_so_excel_does_not_mangle_chinese(self, client):
response = await client.get("/api/monitor/export", params={"kind": "comments"})
assert response.status_code == 200
assert response.content.startswith(b"\xef\xbb\xbf")
assert "attachment" in response.headers["content-disposition"]
text = response.content.decode("utf-8-sig")
rows = list(csv.DictReader(io.StringIO(text)))
assert len(rows) == 4
assert rows[0]["所属作品"] in ("作品甲", "作品乙")
@pytest.mark.asyncio
async def test_notes_export(self, client):
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert {r["作品ID"] for r in rows} == {"note-a", "note-b"}
@pytest.mark.asyncio
async def test_xlsx_is_a_readable_workbook(self, client):
from openpyxl import load_workbook
response = await client.get(
"/api/monitor/export", params={"kind": "comments", "format": "xlsx"}
)
assert response.status_code == 200
workbook = load_workbook(io.BytesIO(response.content))
sheet = workbook.active
assert sheet.max_row == 5 # header + four comments
assert sheet.cell(row=1, column=1).value == "所属作品"
@pytest.mark.asyncio
async def test_report_export(self, client):
response = await client.get(
"/api/monitor/export",
params={"kind": "report", "days": 3},
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert len(rows) == 3
assert "日期" in rows[0]
@pytest.mark.asyncio
async def test_unknown_kind_and_format_are_rejected(self, client):
assert (
await client.get("/api/monitor/export", params={"kind": "nope"})
).status_code == 400
assert (
await client.get("/api/monitor/export", params={"kind": "notes", "format": "pdf"})
).status_code == 400
@pytest.mark.asyncio
async def test_empty_selection_is_a_404_not_an_empty_file(self, client):
"""An empty download looks like a bug; say so instead."""
response = await client.get(
"/api/monitor/export", params={"kind": "comments", "note_id": "no-such-note"}
)
assert response.status_code == 404