Files
MediaCrawler/tests/test_monitor_report.py
T
butubb e77e5e2f15
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
fix(report): 空的任务集合被当成了「不限制平台」,导致报表串平台数据
现象:切到抖音,报表里显示的是小红书的数据。

根因是报告聚合里的一行真值判断:

    scope = list(task_ids) if task_ids else None

空列表是假值,而空列表在这里的含义是「这个平台一个任务都没有」,不是「不限制平台」。
于是 platform=dy 且抖音还没有任务时,_resolve_scope 返回的 [] 被翻译成了 None,
聚合范围从「抖音的任务」变成了**全部任务** —— 小红书的数字就这么显示在了抖音页面上。
顺带 task_ids 也回成 None,界面会显示成「全部任务」。

改成 `is not None`。空列表进去就让 in_([]) 恒假,结果为空,这才是对的。

排查时把所有同类写法过了一遍,只有这一处错,其余(service.py 的 10 处作用域judgement、
_resolve_scope、export)用的都是 `is not None`。

测试:新增两条,并且**验证过它们在修复前会红**(失败信息就是 assert 42 == 0 ——
查一个没有任何任务的平台,却返回了小红书那条作品的 42 个赞)。

同时修掉一条空跑的测试:test_a_platform_with_no_tasks_yields_empty_not_everything
原先种了任务却没有作品/指标数据,于是过滤生效与否结果都是 0,什么都测不出来 ——
这正是这个 bug 能活下来的原因。现在它会真的塞一条作品+快照进去,并在末尾断言
「小红书自己的报表看得到那条数据」,用来证明前面那两个 0 是过滤出来的而不是没数据。
2026-10-10 15:00:12 +08:00

324 lines
12 KiB
Python

# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/tests/test_monitor_report.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""Tests for the cross-task report aggregation.
The interaction delta is the part that is easy to get subtly wrong, so it is
covered directly against the pure aggregation function.
"""
from datetime import date, datetime
import pytest
import pytest_asyncio
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine
from sqlalchemy.pool import StaticPool
from api.monitor.models import (
MODE_CREATOR,
MonitorBase,
MonitorComment,
MonitorNote,
MonitorNoteMetric,
MonitorTask,
)
from api.monitor.report import build_report, compute_daily_rows, day_bounds, iter_days
def _ms(year: int, month: int, day: int, hour: int = 12) -> int:
return int(datetime(year, month, day, hour).timestamp() * 1000)
def _metrics(liked=0, comment=0, collected=0, share=0):
"""All four metrics default to parsed values; pass None to simulate a
platform value we could not parse."""
return {
"liked_count": liked,
"comment_count": comment,
"collected_count": collected,
"share_count": share,
}
class TestDayHelpers:
def test_day_bounds_cover_the_whole_local_day(self):
start, end = day_bounds(date(2026, 1, 10))
assert start < _ms(2026, 1, 10, 0) or start == _ms(2026, 1, 10, 0)
assert end > _ms(2026, 1, 10, 23)
def test_iter_days_is_inclusive(self):
days = iter_days(date(2026, 1, 10), date(2026, 1, 12))
assert days == [date(2026, 1, 10), date(2026, 1, 11), date(2026, 1, 12)]
class TestInteractionDelta:
def test_note_first_seen_counts_all_of_its_value(self):
"""A brand-new note has no earlier baseline, so it starts from zero."""
day = date(2026, 1, 10)
series = {"n1": [(_ms(2026, 1, 10, 10), _metrics(liked=100, comment=5))]}
rows = compute_daily_rows(series, {}, {}, [day])
assert rows[0]["liked_count_delta"] == 100
assert rows[0]["comment_count_delta"] == 5
def test_growth_is_split_across_days(self):
series = {
"n1": [
(_ms(2026, 1, 10, 10), _metrics(liked=100)),
(_ms(2026, 1, 11, 10), _metrics(liked=300)),
]
}
rows = compute_daily_rows(series, {}, {}, [date(2026, 1, 10), date(2026, 1, 11)])
# Day 1: 0 -> 100. Day 2: 100 -> 300.
assert [row["liked_count_delta"] for row in rows] == [100, 200]
def test_day_without_a_snapshot_reports_no_growth(self):
series = {
"n1": [
(_ms(2026, 1, 10, 10), _metrics(liked=100)),
(_ms(2026, 1, 12, 10), _metrics(liked=400)),
]
}
days = [date(2026, 1, 10), date(2026, 1, 11), date(2026, 1, 12)]
rows = compute_daily_rows(series, {}, {}, days)
# The note was not crawled on the 11th, so nothing is claimed for it.
assert [row["liked_count_delta"] for row in rows] == [100, 0, 300]
def test_deltas_aggregate_across_notes(self):
series = {
"n1": [
(_ms(2026, 1, 10, 10), _metrics(liked=100)),
(_ms(2026, 1, 11, 10), _metrics(liked=150)),
],
"n2": [
(_ms(2026, 1, 10, 10), _metrics(liked=10)),
(_ms(2026, 1, 11, 10), _metrics(liked=40)),
],
}
rows = compute_daily_rows(series, {}, {}, [date(2026, 1, 10), date(2026, 1, 11)])
assert [row["liked_count_delta"] for row in rows] == [110, 80]
def test_unparseable_metric_names_the_offending_field(self):
"""A NULL count makes the delta unknown; it must not be reported as 0."""
series = {
"n1": [
(_ms(2026, 1, 10, 10), _metrics(liked=100, comment=None)),
(_ms(2026, 1, 11, 10), _metrics(liked=200, comment=None)),
]
}
rows = compute_daily_rows(series, {}, {}, [date(2026, 1, 11)])
# Naming the field is actionable; a bare boolean is not.
assert rows[0]["partial_metrics"] == ["comment_count"]
# The parseable metric is still summed correctly.
assert rows[0]["liked_count_delta"] == 100
def test_unknown_value_only_taints_the_days_it_touches(self):
series = {
"n1": [
(_ms(2026, 1, 10, 10), _metrics(liked=None)),
(_ms(2026, 1, 11, 10), _metrics(liked=50)),
(_ms(2026, 1, 12, 10), _metrics(liked=90)),
]
}
days = [date(2026, 1, 10), date(2026, 1, 11), date(2026, 1, 12)]
rows = compute_daily_rows(series, {}, {}, days)
# Day 12 compares two known values, so it is clean.
assert [row["partial_metrics"] for row in rows] == [
["liked_count"],
["liked_count"],
[],
]
assert rows[2]["liked_count_delta"] == 40
def test_new_content_counts_come_from_the_day_maps(self):
rows = compute_daily_rows(
{},
{date(2026, 1, 10): 3},
{date(2026, 1, 10): 7},
[date(2026, 1, 10), date(2026, 1, 11)],
)
assert rows[0]["new_notes"] == 3
assert rows[0]["new_comments"] == 7
assert rows[1]["new_notes"] == 0
# --------------------------------------------------------------------------
# DB-backed report + task filtering
# --------------------------------------------------------------------------
@pytest_asyncio.fixture
async def db():
engine = create_async_engine("sqlite+aiosqlite://", poolclass=StaticPool)
async with engine.begin() as conn:
await conn.run_sync(MonitorBase.metadata.create_all)
factory = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False)
async with factory() as session:
yield session
await engine.dispose()
async def _seed_task(db: AsyncSession, name: str, platform: str = "xhs") -> MonitorTask:
task = MonitorTask(
name=name, platform=platform, mode=MODE_CREATOR, enabled=True,
interval_minutes=60, max_notes_count=20, enable_comments=True,
max_comments_count=50, run_timeout_seconds=3600,
notify_enabled=False, created_at=0, updated_at=0,
)
db.add(task)
await db.flush()
return task
async def _seed_note_with_metrics(
db: AsyncSession, task: MonitorTask, note_id: str, samples
) -> None:
db.add(
MonitorNote(
task_id=task.id, note_id=note_id, title=note_id, note_url="",
cover="", creator_hash="", source_kind="", published_at=None,
first_seen_run_id=1, first_seen_at=samples[0][0],
last_seen_run_id=len(samples), last_seen_at=samples[-1][0],
)
)
for run_id, (ts, liked) in enumerate(samples, start=1):
db.add(
MonitorNoteMetric(
task_id=task.id, note_id=note_id, run_id=run_id, captured_at=ts,
liked_count=liked, comment_count=0, collected_count=0, share_count=0,
raw_liked_count=str(liked), raw_comment_count="0",
raw_collected_count="0", raw_share_count="0",
)
)
class TestBuildReport:
@pytest.mark.asyncio
async def test_totals_and_rows(self, db):
task = await _seed_task(db, "t1")
await _seed_note_with_metrics(
db, task, "n1",
[(_ms(2026, 1, 10, 10), 100), (_ms(2026, 1, 11, 10), 250)],
)
await db.commit()
result = await build_report(db, [task.id], date(2026, 1, 10), date(2026, 1, 11))
assert result["totals"]["liked_count_delta"] == 250
assert len(result["rows"]) == 2
assert result["note_count"] == 1
@pytest.mark.asyncio
async def test_task_selection_isolates_the_report(self, db):
"""The whole point: a report for a chosen subset must exclude the rest."""
kept = await _seed_task(db, "kept")
other = await _seed_task(db, "other")
await _seed_note_with_metrics(db, kept, "n1", [(_ms(2026, 1, 10, 10), 100)])
await _seed_note_with_metrics(db, other, "n2", [(_ms(2026, 1, 10, 10), 999)])
await db.commit()
only_kept = await build_report(db, [kept.id], date(2026, 1, 10), date(2026, 1, 10))
assert only_kept["totals"]["liked_count_delta"] == 100
assert only_kept["note_count"] == 1
both = await build_report(db, [kept.id, other.id], date(2026, 1, 10), date(2026, 1, 10))
assert both["totals"]["liked_count_delta"] == 1099
@pytest.mark.asyncio
async def test_no_task_filter_covers_everything(self, db):
first = await _seed_task(db, "a")
second = await _seed_task(db, "b")
await _seed_note_with_metrics(db, first, "n1", [(_ms(2026, 1, 10, 10), 10)])
await _seed_note_with_metrics(db, second, "n2", [(_ms(2026, 1, 10, 10), 20)])
await db.commit()
result = await build_report(db, None, date(2026, 1, 10), date(2026, 1, 10))
assert result["totals"]["liked_count_delta"] == 30
assert result["task_ids"] is None
@pytest.mark.asyncio
async def test_an_empty_selection_is_not_the_same_as_no_filter(self, db):
"""空列表 ≠ 不限制。
``_resolve_scope`` 在「这个平台一个任务都没有」时返回**空列表**。如果按真值
处理(``if task_ids``),报表就会退化成「不限制平台」,把**所有**任务的数据
聚合进来 —— 现象就是切到抖音,报表里却全是小红书的数据。
"""
other = await _seed_task(db, "xhs task", platform="xhs")
await _seed_note_with_metrics(db, other, "n1", [(_ms(2026, 1, 10, 10), 999)])
await db.commit()
result = await build_report(db, [], date(2026, 1, 10), date(2026, 1, 10))
assert result["totals"]["liked_count_delta"] == 0
assert result["note_count"] == 0
# 空列表要原样透出去;None 在 API 里的意思是「全部任务」,两者不能混。
assert result["task_ids"] == []
@pytest.mark.asyncio
async def test_baseline_from_before_the_range_is_used(self, db):
"""Growth is measured against the last value before the window opens."""
task = await _seed_task(db, "t")
await _seed_note_with_metrics(
db, task, "n1",
[(_ms(2026, 1, 5, 10), 1000), (_ms(2026, 1, 10, 10), 1050)],
)
await db.commit()
# Report only for the 10th: the delta must be 50, not 1050.
result = await build_report(db, [task.id], date(2026, 1, 10), date(2026, 1, 10))
assert result["totals"]["liked_count_delta"] == 50
@pytest.mark.asyncio
async def test_empty_range_returns_zeroed_rows(self, db):
result = await build_report(db, None, date(2026, 2, 1), date(2026, 2, 3))
assert len(result["rows"]) == 3
assert result["totals"]["liked_count_delta"] == 0
assert result["totals"]["new_notes"] == 0
@pytest.mark.asyncio
async def test_new_comments_are_counted_by_first_seen_day(self, db):
task = await _seed_task(db, "t")
db.add(
MonitorComment(
task_id=task.id, note_id="n1", comment_id="c1", content="x",
nickname="u", creator_hash="h", create_time=_ms(2026, 1, 9),
like_count=0, sub_comment_count=0, parent_comment_id="",
first_seen_run_id=1, first_seen_at=_ms(2026, 1, 10, 10),
)
)
await db.commit()
result = await build_report(db, [task.id], date(2026, 1, 10), date(2026, 1, 10))
assert result["totals"]["new_comments"] == 1