Files
MediaCrawler/api/monitor/report.py
T
butubb 4e60524f37
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
feat: 监控面板 / 登录鉴权 / 多平台切换 / MySQL
在上游 MediaCrawler 之上新增一层:

- 监控层 api/monitor/ —— 多博主/多笔记的定时采集、指标快照差分、报表、
  企业微信通知。每轮采集写入独立目录,差分才成立。
- WebUI 登录鉴权 api/auth.py —— PBKDF2 口令 + 服务端会话,/api 全接口防护。
  WebSocket 单独加依赖:BaseHTTPMiddleware 对 ws 作用域直接放行,覆盖不到。
- 全局平台切换 + 能力矩阵 —— 如实区分「爬虫模块支持」与「监控层已接线」,
  未接通的平台直接拒绝建任务,而不是静默跑空。
- 监控库改用 MySQL 5.7(可回退 SQLite 供测试):逐表强制 utf8mb4
  (服务端与库默认都是 latin1),启动校验所连 schema 以防写错库,
  连接池 recycle + pre_ping 应对 MySQL 的 8 小时空闲断连。

修复上游缺陷:

- xhs/core.py: 主页抓取失败会跳掉整个博主,导致一条作品都抓不到,
  而那份资料只喂给一个空函数。改为尽力而为,失败不中断。
- xhs/login.py: cookie 登录只注入 web_session,冷启动签名会失败。
  新增 INJECT_ALL_COOKIES 开关(默认关闭,原有行为不变)。
- requirements.txt: 补上 websockets。它在上游 pyproject.toml 里有声明、
  这里漏了,导致 uvicorn 没有 WebSocket 能力,实时日志流从未工作。

改动过的上游文件清单及合并方式见 UPSTREAM.md。

测试:492 passed(另有 1 个既有的 Windows/gbk 上游测试失败,与本改动无关)
2026-10-07 09:58:40 +08:00

204 lines
7.9 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/report.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""Cross-task reporting: what grew, and what is new, over a date range.
Two families of numbers that answer different questions and are therefore kept
as separate columns:
* **互动增量** — Σ(current − previous) across the selected notes. "How many likes
did this set of notes gain?"
* **新增内容** — count of newly discovered notes and comments. "How much new
material showed up?"
The per-day interaction delta is defined as *last value on the day* minus *last
value before the day* (0 when the note was first seen on that day). That keeps
growth from a note's first observation counted once, rather than smeared across
every later day.
Aggregation runs in Python over the snapshots rather than as one large SQL
query: the per-note-per-day baseline lookup is a windowed operation that SQLite
expresses awkwardly, and the row counts here are small enough that clarity is
worth more than the query planner.
"""
from bisect import bisect_right
from datetime import date, datetime, time, timedelta
from typing import Any, Dict, Iterable, List, Optional, Sequence
from sqlalchemy import select
from sqlalchemy.ext.asyncio import AsyncSession
from .models import MonitorComment, MonitorNote, MonitorNoteMetric
METRIC_FIELDS = ("liked_count", "comment_count", "collected_count", "share_count")
METRIC_LABELS = {
"liked_count": "点赞",
"comment_count": "评论",
"collected_count": "收藏",
"share_count": "分享",
}
def day_bounds(day: date) -> tuple[int, int]:
"""Inclusive epoch-millisecond bounds for a local calendar day."""
start = datetime.combine(day, time.min)
end = datetime.combine(day, time.max)
return int(start.timestamp() * 1000), int(end.timestamp() * 1000)
def iter_days(start: date, end: date) -> List[date]:
days = []
cursor = start
while cursor <= end:
days.append(cursor)
cursor += timedelta(days=1)
return days
def compute_daily_rows(
series_by_note: Dict[str, List[tuple[int, Dict[str, Optional[int]]]]],
notes_per_day: Dict[date, int],
comments_per_day: Dict[date, int],
days: Sequence[date],
) -> List[Dict[str, Any]]:
"""Pure aggregation. ``series_by_note`` must be sorted by timestamp ascending."""
prepared = {note_id: ([ts for ts, _ in points], points) for note_id, points in series_by_note.items()}
rows: List[Dict[str, Any]] = []
for day in days:
day_start, day_end = day_bounds(day)
totals = {field: 0 for field in METRIC_FIELDS}
# Records *which* metric could not be compared, not just that something
# could not. A blanket flag loses all value the moment one permanently
# unparseable field makes every row "incomplete".
partial_metrics: set[str] = set()
for times, points in prepared.values():
end_index = bisect_right(times, day_end) - 1
if end_index < 0:
# Not yet tracked on this day.
continue
end_values = points[end_index][1]
start_index = bisect_right(times, day_start - 1) - 1
# No earlier snapshot means the note first appeared in this window,
# so it starts from zero -- all of its count is genuinely new.
start_values = (
points[start_index][1] if start_index >= 0 else {f: 0 for f in METRIC_FIELDS}
)
for field in METRIC_FIELDS:
end_value, start_value = end_values.get(field), start_values.get(field)
if end_value is None or start_value is None:
# An unparseable count on either side makes the delta unknown;
# skipping beats reporting a fabricated number.
partial_metrics.add(field)
continue
totals[field] += end_value - start_value
row: Dict[str, Any] = {
"date": day.isoformat(),
"new_notes": notes_per_day.get(day, 0),
"new_comments": comments_per_day.get(day, 0),
"partial_metrics": sorted(partial_metrics),
}
row.update({f"{field}_delta": value for field, value in totals.items()})
rows.append(row)
return rows
async def build_report(
session: AsyncSession,
task_ids: Optional[Iterable[int]],
start_day: date,
end_day: date,
) -> Dict[str, Any]:
"""Daily rows plus totals for the selected tasks over the given date range."""
start_ms, _ = day_bounds(start_day)
_, end_ms = day_bounds(end_day)
scope = list(task_ids) if task_ids else None
days = iter_days(start_day, end_day)
# Fetch every snapshot up to the range end: the delta on the first day needs
# the last value from *before* the range, so a lower bound would be wrong.
metric_stmt = select(MonitorNoteMetric).where(MonitorNoteMetric.captured_at <= end_ms)
if scope is not None:
metric_stmt = metric_stmt.where(MonitorNoteMetric.task_id.in_(scope))
metric_stmt = metric_stmt.order_by(MonitorNoteMetric.note_id, MonitorNoteMetric.run_id)
series_by_note: Dict[str, List[tuple[int, Dict[str, Optional[int]]]]] = {}
included_note_ids: set[str] = set()
for snapshot in (await session.scalars(metric_stmt)).all():
included_note_ids.add(snapshot.note_id)
series_by_note.setdefault(snapshot.note_id, []).append(
(
snapshot.captured_at,
{field: getattr(snapshot, field) for field in METRIC_FIELDS},
)
)
note_stmt = select(MonitorNote.first_seen_at).where(
MonitorNote.first_seen_at >= start_ms, MonitorNote.first_seen_at <= end_ms
)
if scope is not None:
note_stmt = note_stmt.where(MonitorNote.task_id.in_(scope))
comment_stmt = select(MonitorComment.first_seen_at).where(
MonitorComment.first_seen_at >= start_ms, MonitorComment.first_seen_at <= end_ms
)
if scope is not None:
comment_stmt = comment_stmt.where(MonitorComment.task_id.in_(scope))
notes_per_day = _count_by_day((await session.scalars(note_stmt)).all())
comments_per_day = _count_by_day((await session.scalars(comment_stmt)).all())
rows = compute_daily_rows(series_by_note, notes_per_day, comments_per_day, days)
totals = {
"new_notes": sum(row["new_notes"] for row in rows),
"new_comments": sum(row["new_comments"] for row in rows),
}
for field in METRIC_FIELDS:
totals[f"{field}_delta"] = sum(row[f"{field}_delta"] for row in rows)
return {
"start_date": start_day.isoformat(),
"end_date": end_day.isoformat(),
"task_ids": scope,
"rows": rows,
"totals": totals,
"note_count": len(included_note_ids),
"has_partial_data": any(row["partial_metrics"] for row in rows),
"partial_metrics": sorted({field for row in rows for field in row["partial_metrics"]}),
"metric_labels": METRIC_LABELS,
}
def _count_by_day(timestamps: Iterable[Optional[int]]) -> Dict[date, int]:
counts: Dict[date, int] = {}
for ts in timestamps:
if ts is None:
continue
day = datetime.fromtimestamp(ts / 1000).date()
counts[day] = counts.get(day, 0) + 1
return counts