Files
MediaCrawler/api/monitor/report.py
T
butubb e77e5e2f15
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
fix(report): 空的任务集合被当成了「不限制平台」,导致报表串平台数据
现象:切到抖音,报表里显示的是小红书的数据。

根因是报告聚合里的一行真值判断:

    scope = list(task_ids) if task_ids else None

空列表是假值,而空列表在这里的含义是「这个平台一个任务都没有」,不是「不限制平台」。
于是 platform=dy 且抖音还没有任务时,_resolve_scope 返回的 [] 被翻译成了 None,
聚合范围从「抖音的任务」变成了**全部任务** —— 小红书的数字就这么显示在了抖音页面上。
顺带 task_ids 也回成 None,界面会显示成「全部任务」。

改成 `is not None`。空列表进去就让 in_([]) 恒假,结果为空,这才是对的。

排查时把所有同类写法过了一遍,只有这一处错,其余(service.py 的 10 处作用域judgement、
_resolve_scope、export)用的都是 `is not None`。

测试:新增两条,并且**验证过它们在修复前会红**(失败信息就是 assert 42 == 0 ——
查一个没有任何任务的平台,却返回了小红书那条作品的 42 个赞)。

同时修掉一条空跑的测试:test_a_platform_with_no_tasks_yields_empty_not_everything
原先种了任务却没有作品/指标数据,于是过滤生效与否结果都是 0,什么都测不出来 ——
这正是这个 bug 能活下来的原因。现在它会真的塞一条作品+快照进去,并在末尾断言
「小红书自己的报表看得到那条数据」,用来证明前面那两个 0 是过滤出来的而不是没数据。
2026-10-10 15:00:12 +08:00

208 lines
8.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/report.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""Cross-task reporting: what grew, and what is new, over a date range.
Two families of numbers that answer different questions and are therefore kept
as separate columns:
* **互动增量** — Σ(current − previous) across the selected notes. "How many likes
did this set of notes gain?"
* **新增内容** — count of newly discovered notes and comments. "How much new
material showed up?"
The per-day interaction delta is defined as *last value on the day* minus *last
value before the day* (0 when the note was first seen on that day). That keeps
growth from a note's first observation counted once, rather than smeared across
every later day.
Aggregation runs in Python over the snapshots rather than as one large SQL
query: the per-note-per-day baseline lookup is a windowed operation that SQLite
expresses awkwardly, and the row counts here are small enough that clarity is
worth more than the query planner.
"""
from bisect import bisect_right
from datetime import date, datetime, time, timedelta
from typing import Any, Dict, Iterable, List, Optional, Sequence
from sqlalchemy import select
from sqlalchemy.ext.asyncio import AsyncSession
from .models import MonitorComment, MonitorNote, MonitorNoteMetric
METRIC_FIELDS = ("liked_count", "comment_count", "collected_count", "share_count")
METRIC_LABELS = {
"liked_count": "点赞",
"comment_count": "评论",
"collected_count": "收藏",
"share_count": "分享",
}
def day_bounds(day: date) -> tuple[int, int]:
"""Inclusive epoch-millisecond bounds for a local calendar day."""
start = datetime.combine(day, time.min)
end = datetime.combine(day, time.max)
return int(start.timestamp() * 1000), int(end.timestamp() * 1000)
def iter_days(start: date, end: date) -> List[date]:
days = []
cursor = start
while cursor <= end:
days.append(cursor)
cursor += timedelta(days=1)
return days
def compute_daily_rows(
series_by_note: Dict[str, List[tuple[int, Dict[str, Optional[int]]]]],
notes_per_day: Dict[date, int],
comments_per_day: Dict[date, int],
days: Sequence[date],
) -> List[Dict[str, Any]]:
"""Pure aggregation. ``series_by_note`` must be sorted by timestamp ascending."""
prepared = {note_id: ([ts for ts, _ in points], points) for note_id, points in series_by_note.items()}
rows: List[Dict[str, Any]] = []
for day in days:
day_start, day_end = day_bounds(day)
totals = {field: 0 for field in METRIC_FIELDS}
# Records *which* metric could not be compared, not just that something
# could not. A blanket flag loses all value the moment one permanently
# unparseable field makes every row "incomplete".
partial_metrics: set[str] = set()
for times, points in prepared.values():
end_index = bisect_right(times, day_end) - 1
if end_index < 0:
# Not yet tracked on this day.
continue
end_values = points[end_index][1]
start_index = bisect_right(times, day_start - 1) - 1
# No earlier snapshot means the note first appeared in this window,
# so it starts from zero -- all of its count is genuinely new.
start_values = (
points[start_index][1] if start_index >= 0 else {f: 0 for f in METRIC_FIELDS}
)
for field in METRIC_FIELDS:
end_value, start_value = end_values.get(field), start_values.get(field)
if end_value is None or start_value is None:
# An unparseable count on either side makes the delta unknown;
# skipping beats reporting a fabricated number.
partial_metrics.add(field)
continue
totals[field] += end_value - start_value
row: Dict[str, Any] = {
"date": day.isoformat(),
"new_notes": notes_per_day.get(day, 0),
"new_comments": comments_per_day.get(day, 0),
"partial_metrics": sorted(partial_metrics),
}
row.update({f"{field}_delta": value for field, value in totals.items()})
rows.append(row)
return rows
async def build_report(
session: AsyncSession,
task_ids: Optional[Iterable[int]],
start_day: date,
end_day: date,
) -> Dict[str, Any]:
"""Daily rows plus totals for the selected tasks over the given date range."""
start_ms, _ = day_bounds(start_day)
_, end_ms = day_bounds(end_day)
# 必须是 `is not None`,不能写 `if task_ids` —— **空列表是假值**,而空列表在这里
# 的含义是「这个平台一个任务都没有」,不是「不限制平台」。用真值判断的话,
# 切到一个还没有任务的平台,报表会把**所有**任务的数据聚合出来(看起来就是
# 「抖音的报表里全是小红书的数据」)。
scope = list(task_ids) if task_ids is not None else None
days = iter_days(start_day, end_day)
# Fetch every snapshot up to the range end: the delta on the first day needs
# the last value from *before* the range, so a lower bound would be wrong.
metric_stmt = select(MonitorNoteMetric).where(MonitorNoteMetric.captured_at <= end_ms)
if scope is not None:
metric_stmt = metric_stmt.where(MonitorNoteMetric.task_id.in_(scope))
metric_stmt = metric_stmt.order_by(MonitorNoteMetric.note_id, MonitorNoteMetric.run_id)
series_by_note: Dict[str, List[tuple[int, Dict[str, Optional[int]]]]] = {}
included_note_ids: set[str] = set()
for snapshot in (await session.scalars(metric_stmt)).all():
included_note_ids.add(snapshot.note_id)
series_by_note.setdefault(snapshot.note_id, []).append(
(
snapshot.captured_at,
{field: getattr(snapshot, field) for field in METRIC_FIELDS},
)
)
note_stmt = select(MonitorNote.first_seen_at).where(
MonitorNote.first_seen_at >= start_ms, MonitorNote.first_seen_at <= end_ms
)
if scope is not None:
note_stmt = note_stmt.where(MonitorNote.task_id.in_(scope))
comment_stmt = select(MonitorComment.first_seen_at).where(
MonitorComment.first_seen_at >= start_ms, MonitorComment.first_seen_at <= end_ms
)
if scope is not None:
comment_stmt = comment_stmt.where(MonitorComment.task_id.in_(scope))
notes_per_day = _count_by_day((await session.scalars(note_stmt)).all())
comments_per_day = _count_by_day((await session.scalars(comment_stmt)).all())
rows = compute_daily_rows(series_by_note, notes_per_day, comments_per_day, days)
totals = {
"new_notes": sum(row["new_notes"] for row in rows),
"new_comments": sum(row["new_comments"] for row in rows),
}
for field in METRIC_FIELDS:
totals[f"{field}_delta"] = sum(row[f"{field}_delta"] for row in rows)
return {
"start_date": start_day.isoformat(),
"end_date": end_day.isoformat(),
"task_ids": scope,
"rows": rows,
"totals": totals,
"note_count": len(included_note_ids),
"has_partial_data": any(row["partial_metrics"] for row in rows),
"partial_metrics": sorted({field for row in rows for field in row["partial_metrics"]}),
"metric_labels": METRIC_LABELS,
}
def _count_by_day(timestamps: Iterable[Optional[int]]) -> Dict[date, int]:
counts: Dict[date, int] = {}
for ts in timestamps:
if ts is None:
continue
day = datetime.fromtimestamp(ts / 1000).date()
counts[day] = counts.get(day, 0) + 1
return counts