Files
MediaCrawler/tests/test_monitor_comments.py
T
butubb 3486c7f524
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
feat(monitor): 作品栏和评论栏显示作品的发布日期
需求:抖音和小红书都要能看到作品的发布日期。

`MonitorNote.published_at` 其实**一直在库里**(ingest 早就按平台取:小红书 time、
抖音 create_time),只是从来没往 API 和界面上透 —— 后端序列化没这个键,前端类型里
也没有,所以界面上只有「首次发现」。

* 后端:list_notes 的序列化补上 published_at;_note_meta_map 也带上,于是
  list_comments 多一个 note_published_at,分组接口的桶多一个 published_at。
* 前端:NotesTable 新增「发布日期」列(要让分组表头的 colspan 从 +4 变 +5);
  评论栏作品那一层在标题旁显示日期 —— 同名作品不少,日期能帮着认。
* 新增 formatDate:发布日期问的是「哪一天发的」,绝对日期比「3天前」好认,也不会
  每天看都在变。具体到分钟的版本放在 title 里,悬停可见。

刻意和「首次发现」分开:前者是作者发布的那天,后者是我们第一次看到它的那天。把一个
早就存在的作品加进监控时,两者能差好几个月 —— 测试里就用不同的值把这两者钉住。

顺带修正一处过时注释:前端类型里还写着 creator_name 是「已脱敏的昵称」,
脱敏已经在上一个提交里关掉了(config.MASK_NICKNAME)。

测试 +2:桶要带发布日期;作品列表接口要带,且它不等于 first_seen_at。
2026-10-10 16:06:22 +08:00

295 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/tests/test_monitor_comments.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""Comment note-association, grouping, and the export endpoint."""
import csv
import io
import httpx
import pytest
import pytest_asyncio
from api.main import app
from api.monitor import db as monitor_db
from api.monitor.models import (
MODE_CREATOR,
MonitorComment,
MonitorNote,
MonitorTask,
)
TASK_NAME = "评论归属测试"
# 作品的发布时间。和 first_seen_at(我们第一次看到它)刻意取不同的值 —— 两者混成
# 一个概念是最容易犯的错。
PUBLISHED_A = 1_699_000_000_000
PUBLISHED_B = 1_699_100_000_000
async def _seed():
"""Two works; three comments on the first, one on the second."""
async with monitor_db.get_session() as session:
task = MonitorTask(
name=TASK_NAME, platform="xhs", mode=MODE_CREATOR, enabled=True,
interval_minutes=60, max_notes_count=20, enable_comments=True,
max_comments_count=50, run_timeout_seconds=3600,
notify_enabled=False, created_at=0, updated_at=0,
)
session.add(task)
await session.flush()
# 两个作品**属于不同的博主** —— 评论流最外层按创作者分组,同一个人就没得测了。
for note_id, title, creator_hash, creator_name, published_at in (
("note-a", "作品甲", "hash-a", "博主甲", PUBLISHED_A),
("note-b", "作品乙", "hash-b", "博主乙", PUBLISHED_B),
):
session.add(
MonitorNote(
task_id=task.id, note_id=note_id, title=title,
note_url=f"https://www.xiaohongshu.com/explore/{note_id}",
cover=f"https://img/{note_id}.jpg", creator_hash=creator_hash,
creator_name=creator_name,
source_kind="video", published_at=published_at,
first_seen_run_id=1, first_seen_at=1_700_000_000_000,
last_seen_run_id=1, last_seen_at=1_700_000_000_000,
)
)
# note-a has three comments, note-b has one.
plan = [
("c1", "note-a", 1_700_000_001_000),
("c2", "note-a", 1_700_000_002_000),
("c3", "note-a", 1_700_000_003_000),
("c4", "note-b", 1_700_000_004_000),
]
for comment_id, note_id, seen_at in plan:
session.add(
MonitorComment(
task_id=task.id, note_id=note_id, comment_id=comment_id,
content=f"内容-{comment_id}", nickname="u***r", creator_hash="h",
create_time=seen_at, like_count=1, sub_comment_count=0,
parent_comment_id="", first_seen_run_id=1, first_seen_at=seen_at,
)
)
return task.id
@pytest_asyncio.fixture
async def client(tmp_path):
monitor_db.set_sqlite_path(tmp_path / "monitor.db")
await monitor_db.init_db()
await _seed()
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url="http://test") as http_client:
yield http_client
await monitor_db.dispose_engine()
class TestCommentsCarryTheirNote:
@pytest.mark.asyncio
async def test_each_comment_names_its_work(self, client):
"""A bare note_id is unreadable -- the title is the whole point."""
response = await client.get("/api/monitor/comments")
assert response.status_code == 200
comments = response.json()["comments"]
assert len(comments) == 4
by_id = {c["comment_id"]: c for c in comments}
assert by_id["c1"]["note_title"] == "作品甲"
assert by_id["c1"]["note_url"].endswith("note-a")
assert by_id["c1"]["note_cover"].endswith("note-a.jpg")
assert by_id["c4"]["note_title"] == "作品乙"
@pytest.mark.asyncio
async def test_note_id_filters_the_stream(self, client):
response = await client.get("/api/monitor/comments", params={"note_id": "note-a"})
comments = response.json()["comments"]
assert {c["comment_id"] for c in comments} == {"c1", "c2", "c3"}
class TestGroupByNote:
@pytest.mark.asyncio
async def test_groups_bucket_by_work(self, client):
response = await client.get("/api/monitor/comments", params={"group_by": "note"})
body = response.json()
assert "groups" in body
assert body["total"] == 4
groups = {g["note_id"]: g for g in body["groups"]}
assert set(groups) == {"note-a", "note-b"}
assert len(groups["note-a"]["comments"]) == 3
assert len(groups["note-b"]["comments"]) == 1
assert groups["note-a"]["note_title"] == "作品甲"
@pytest.mark.asyncio
async def test_newest_group_comes_first(self, client):
"""The UI expands the first group by default, so it must be the newest."""
response = await client.get("/api/monitor/comments", params={"group_by": "note"})
groups = response.json()["groups"]
# note-b's only comment is the most recent overall.
assert groups[0]["note_id"] == "note-b"
@pytest.mark.asyncio
async def test_each_bucket_carries_its_creator(self, client):
"""桶上必须带作品的创作者 —— 评论流最外层就是按它分组的。
少了这两个字段,前端拿到的 creator_hash / creator_name 都是 undefined,
于是所有博主塌成同一个分组、标签回退成「未知博主」:一个人都分不出来。
"""
groups = {
group["note_id"]: group
for group in (
await client.get("/api/monitor/comments", params={"group_by": "note"})
).json()["groups"]
}
assert groups["note-a"]["creator_hash"] == "hash-a"
assert groups["note-a"]["creator_name"] == "博主甲"
assert groups["note-b"]["creator_hash"] == "hash-b"
assert groups["note-b"]["creator_name"] == "博主乙"
# 两个作品的创作者必须真的不同,否则界面上照样分不出来。
assert groups["note-a"]["creator_hash"] != groups["note-b"]["creator_hash"]
@pytest.mark.asyncio
async def test_each_bucket_carries_the_publish_date(self, client):
"""作品那一层要带发布日期:同名作品不少,日期能帮着认。
注意它和 first_seen_at 是两个概念 —— 前者是作者发布的那天,后者是我们第一次
看到它的那天。把老作品加进监控时两者能差好几个月。
"""
groups = {
group["note_id"]: group
for group in (
await client.get("/api/monitor/comments", params={"group_by": "note"})
).json()["groups"]
}
assert groups["note-a"]["published_at"] == PUBLISHED_A
assert groups["note-b"]["published_at"] == PUBLISHED_B
assert groups["note-a"]["published_at"] != groups["note-b"]["published_at"]
@pytest.mark.asyncio
async def test_the_notes_endpoint_exposes_the_publish_date(self, client):
"""作品列表也要带上它 —— 作品栏就是靠这个显示「发布日期」列的。"""
notes = {
note["note_id"]: note
for note in (await client.get("/api/monitor/notes")).json()["notes"]
}
assert notes["note-a"]["published_at"] == PUBLISHED_A
assert notes["note-b"]["published_at"] == PUBLISHED_B
# 和「首次发现」不是同一个值 —— 两者混了的话这个断言会抓到。
assert notes["note-a"]["first_seen_at"] != notes["note-a"]["published_at"]
@pytest.mark.asyncio
async def test_flat_shape_is_unchanged_without_the_flag(self, client):
body = (await client.get("/api/monitor/comments")).json()
assert "comments" in body and "groups" not in body
class TestCommentNoteFilterOptions:
@pytest.mark.asyncio
async def test_options_carry_counts_and_titles(self, client):
response = await client.get("/api/monitor/comment-notes")
assert response.status_code == 200
notes = {n["note_id"]: n for n in response.json()["notes"]}
assert notes["note-a"]["comment_count"] == 3
assert notes["note-b"]["comment_count"] == 1
assert notes["note-a"]["note_title"] == "作品甲"
@pytest.mark.asyncio
async def test_scoped_to_a_task(self, client):
tasks = (await client.get("/api/monitor/tasks")).json()["tasks"]
task_id = tasks[0]["id"]
scoped = await client.get("/api/monitor/comment-notes", params={"task_id": task_id})
assert len(scoped.json()["notes"]) == 2
# A task with no comments yields an empty list, not an error.
other = await client.get("/api/monitor/comment-notes", params={"task_id": 9999})
assert other.json()["notes"] == []
class TestExport:
@pytest.mark.asyncio
async def test_csv_has_a_bom_so_excel_does_not_mangle_chinese(self, client):
response = await client.get("/api/monitor/export", params={"kind": "comments"})
assert response.status_code == 200
assert response.content.startswith(b"\xef\xbb\xbf")
assert "attachment" in response.headers["content-disposition"]
text = response.content.decode("utf-8-sig")
rows = list(csv.DictReader(io.StringIO(text)))
assert len(rows) == 4
assert rows[0]["所属作品"] in ("作品甲", "作品乙")
@pytest.mark.asyncio
async def test_notes_export(self, client):
response = await client.get(
"/api/monitor/export", params={"kind": "notes", "format": "csv"}
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert {r["作品ID"] for r in rows} == {"note-a", "note-b"}
@pytest.mark.asyncio
async def test_xlsx_is_a_readable_workbook(self, client):
from openpyxl import load_workbook
response = await client.get(
"/api/monitor/export", params={"kind": "comments", "format": "xlsx"}
)
assert response.status_code == 200
workbook = load_workbook(io.BytesIO(response.content))
sheet = workbook.active
assert sheet.max_row == 5 # header + four comments
assert sheet.cell(row=1, column=1).value == "所属作品"
@pytest.mark.asyncio
async def test_report_export(self, client):
response = await client.get(
"/api/monitor/export",
params={"kind": "report", "days": 3},
)
rows = list(csv.DictReader(io.StringIO(response.content.decode("utf-8-sig"))))
assert len(rows) == 3
assert "日期" in rows[0]
@pytest.mark.asyncio
async def test_unknown_kind_and_format_are_rejected(self, client):
assert (
await client.get("/api/monitor/export", params={"kind": "nope"})
).status_code == 400
assert (
await client.get("/api/monitor/export", params={"kind": "notes", "format": "pdf"})
).status_code == 400
@pytest.mark.asyncio
async def test_empty_selection_is_a_404_not_an_empty_file(self, client):
"""An empty download looks like a bug; say so instead."""
response = await client.get(
"/api/monitor/export", params={"kind": "comments", "note_id": "no-such-note"}
)
assert response.status_code == 404