在上游 MediaCrawler 之上新增一层: - 监控层 api/monitor/ —— 多博主/多笔记的定时采集、指标快照差分、报表、 企业微信通知。每轮采集写入独立目录,差分才成立。 - WebUI 登录鉴权 api/auth.py —— PBKDF2 口令 + 服务端会话,/api 全接口防护。 WebSocket 单独加依赖:BaseHTTPMiddleware 对 ws 作用域直接放行,覆盖不到。 - 全局平台切换 + 能力矩阵 —— 如实区分「爬虫模块支持」与「监控层已接线」, 未接通的平台直接拒绝建任务,而不是静默跑空。 - 监控库改用 MySQL 5.7(可回退 SQLite 供测试):逐表强制 utf8mb4 (服务端与库默认都是 latin1),启动校验所连 schema 以防写错库, 连接池 recycle + pre_ping 应对 MySQL 的 8 小时空闲断连。 修复上游缺陷: - xhs/core.py: 主页抓取失败会跳掉整个博主,导致一条作品都抓不到, 而那份资料只喂给一个空函数。改为尽力而为,失败不中断。 - xhs/login.py: cookie 登录只注入 web_session,冷启动签名会失败。 新增 INJECT_ALL_COOKIES 开关(默认关闭,原有行为不变)。 - requirements.txt: 补上 websockets。它在上游 pyproject.toml 里有声明、 这里漏了,导致 uvicorn 没有 WebSocket 能力,实时日志流从未工作。 改动过的上游文件清单及合并方式见 UPSTREAM.md。 测试:492 passed(另有 1 个既有的 Windows/gbk 上游测试失败,与本改动无关)
204 lines
7.9 KiB
Python
204 lines
7.9 KiB
Python
# -*- coding: utf-8 -*-
|
||
# Copyright (c) 2025 [email protected]
|
||
#
|
||
# This file is part of MediaCrawler project.
|
||
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/report.py
|
||
# GitHub: https://github.com/NanmiCoder
|
||
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
|
||
#
|
||
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
|
||
# 1. 不得用于任何商业用途。
|
||
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
|
||
# 3. 不得进行大规模爬取或对平台造成运营干扰。
|
||
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
|
||
# 5. 不得用于任何非法或不当的用途。
|
||
#
|
||
# 详细许可条款请参阅项目根目录下的LICENSE文件。
|
||
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
|
||
|
||
"""Cross-task reporting: what grew, and what is new, over a date range.
|
||
|
||
Two families of numbers that answer different questions and are therefore kept
|
||
as separate columns:
|
||
|
||
* **互动增量** — Σ(current − previous) across the selected notes. "How many likes
|
||
did this set of notes gain?"
|
||
* **新增内容** — count of newly discovered notes and comments. "How much new
|
||
material showed up?"
|
||
|
||
The per-day interaction delta is defined as *last value on the day* minus *last
|
||
value before the day* (0 when the note was first seen on that day). That keeps
|
||
growth from a note's first observation counted once, rather than smeared across
|
||
every later day.
|
||
|
||
Aggregation runs in Python over the snapshots rather than as one large SQL
|
||
query: the per-note-per-day baseline lookup is a windowed operation that SQLite
|
||
expresses awkwardly, and the row counts here are small enough that clarity is
|
||
worth more than the query planner.
|
||
"""
|
||
|
||
from bisect import bisect_right
|
||
from datetime import date, datetime, time, timedelta
|
||
from typing import Any, Dict, Iterable, List, Optional, Sequence
|
||
|
||
from sqlalchemy import select
|
||
from sqlalchemy.ext.asyncio import AsyncSession
|
||
|
||
from .models import MonitorComment, MonitorNote, MonitorNoteMetric
|
||
|
||
METRIC_FIELDS = ("liked_count", "comment_count", "collected_count", "share_count")
|
||
|
||
METRIC_LABELS = {
|
||
"liked_count": "点赞",
|
||
"comment_count": "评论",
|
||
"collected_count": "收藏",
|
||
"share_count": "分享",
|
||
}
|
||
|
||
|
||
def day_bounds(day: date) -> tuple[int, int]:
|
||
"""Inclusive epoch-millisecond bounds for a local calendar day."""
|
||
start = datetime.combine(day, time.min)
|
||
end = datetime.combine(day, time.max)
|
||
return int(start.timestamp() * 1000), int(end.timestamp() * 1000)
|
||
|
||
|
||
def iter_days(start: date, end: date) -> List[date]:
|
||
days = []
|
||
cursor = start
|
||
while cursor <= end:
|
||
days.append(cursor)
|
||
cursor += timedelta(days=1)
|
||
return days
|
||
|
||
|
||
def compute_daily_rows(
|
||
series_by_note: Dict[str, List[tuple[int, Dict[str, Optional[int]]]]],
|
||
notes_per_day: Dict[date, int],
|
||
comments_per_day: Dict[date, int],
|
||
days: Sequence[date],
|
||
) -> List[Dict[str, Any]]:
|
||
"""Pure aggregation. ``series_by_note`` must be sorted by timestamp ascending."""
|
||
prepared = {note_id: ([ts for ts, _ in points], points) for note_id, points in series_by_note.items()}
|
||
|
||
rows: List[Dict[str, Any]] = []
|
||
for day in days:
|
||
day_start, day_end = day_bounds(day)
|
||
totals = {field: 0 for field in METRIC_FIELDS}
|
||
# Records *which* metric could not be compared, not just that something
|
||
# could not. A blanket flag loses all value the moment one permanently
|
||
# unparseable field makes every row "incomplete".
|
||
partial_metrics: set[str] = set()
|
||
|
||
for times, points in prepared.values():
|
||
end_index = bisect_right(times, day_end) - 1
|
||
if end_index < 0:
|
||
# Not yet tracked on this day.
|
||
continue
|
||
|
||
end_values = points[end_index][1]
|
||
start_index = bisect_right(times, day_start - 1) - 1
|
||
# No earlier snapshot means the note first appeared in this window,
|
||
# so it starts from zero -- all of its count is genuinely new.
|
||
start_values = (
|
||
points[start_index][1] if start_index >= 0 else {f: 0 for f in METRIC_FIELDS}
|
||
)
|
||
|
||
for field in METRIC_FIELDS:
|
||
end_value, start_value = end_values.get(field), start_values.get(field)
|
||
if end_value is None or start_value is None:
|
||
# An unparseable count on either side makes the delta unknown;
|
||
# skipping beats reporting a fabricated number.
|
||
partial_metrics.add(field)
|
||
continue
|
||
totals[field] += end_value - start_value
|
||
|
||
row: Dict[str, Any] = {
|
||
"date": day.isoformat(),
|
||
"new_notes": notes_per_day.get(day, 0),
|
||
"new_comments": comments_per_day.get(day, 0),
|
||
"partial_metrics": sorted(partial_metrics),
|
||
}
|
||
row.update({f"{field}_delta": value for field, value in totals.items()})
|
||
rows.append(row)
|
||
|
||
return rows
|
||
|
||
|
||
async def build_report(
|
||
session: AsyncSession,
|
||
task_ids: Optional[Iterable[int]],
|
||
start_day: date,
|
||
end_day: date,
|
||
) -> Dict[str, Any]:
|
||
"""Daily rows plus totals for the selected tasks over the given date range."""
|
||
start_ms, _ = day_bounds(start_day)
|
||
_, end_ms = day_bounds(end_day)
|
||
|
||
scope = list(task_ids) if task_ids else None
|
||
days = iter_days(start_day, end_day)
|
||
|
||
# Fetch every snapshot up to the range end: the delta on the first day needs
|
||
# the last value from *before* the range, so a lower bound would be wrong.
|
||
metric_stmt = select(MonitorNoteMetric).where(MonitorNoteMetric.captured_at <= end_ms)
|
||
if scope is not None:
|
||
metric_stmt = metric_stmt.where(MonitorNoteMetric.task_id.in_(scope))
|
||
metric_stmt = metric_stmt.order_by(MonitorNoteMetric.note_id, MonitorNoteMetric.run_id)
|
||
|
||
series_by_note: Dict[str, List[tuple[int, Dict[str, Optional[int]]]]] = {}
|
||
included_note_ids: set[str] = set()
|
||
for snapshot in (await session.scalars(metric_stmt)).all():
|
||
included_note_ids.add(snapshot.note_id)
|
||
series_by_note.setdefault(snapshot.note_id, []).append(
|
||
(
|
||
snapshot.captured_at,
|
||
{field: getattr(snapshot, field) for field in METRIC_FIELDS},
|
||
)
|
||
)
|
||
|
||
note_stmt = select(MonitorNote.first_seen_at).where(
|
||
MonitorNote.first_seen_at >= start_ms, MonitorNote.first_seen_at <= end_ms
|
||
)
|
||
if scope is not None:
|
||
note_stmt = note_stmt.where(MonitorNote.task_id.in_(scope))
|
||
|
||
comment_stmt = select(MonitorComment.first_seen_at).where(
|
||
MonitorComment.first_seen_at >= start_ms, MonitorComment.first_seen_at <= end_ms
|
||
)
|
||
if scope is not None:
|
||
comment_stmt = comment_stmt.where(MonitorComment.task_id.in_(scope))
|
||
|
||
notes_per_day = _count_by_day((await session.scalars(note_stmt)).all())
|
||
comments_per_day = _count_by_day((await session.scalars(comment_stmt)).all())
|
||
|
||
rows = compute_daily_rows(series_by_note, notes_per_day, comments_per_day, days)
|
||
|
||
totals = {
|
||
"new_notes": sum(row["new_notes"] for row in rows),
|
||
"new_comments": sum(row["new_comments"] for row in rows),
|
||
}
|
||
for field in METRIC_FIELDS:
|
||
totals[f"{field}_delta"] = sum(row[f"{field}_delta"] for row in rows)
|
||
|
||
return {
|
||
"start_date": start_day.isoformat(),
|
||
"end_date": end_day.isoformat(),
|
||
"task_ids": scope,
|
||
"rows": rows,
|
||
"totals": totals,
|
||
"note_count": len(included_note_ids),
|
||
"has_partial_data": any(row["partial_metrics"] for row in rows),
|
||
"partial_metrics": sorted({field for row in rows for field in row["partial_metrics"]}),
|
||
"metric_labels": METRIC_LABELS,
|
||
}
|
||
|
||
|
||
def _count_by_day(timestamps: Iterable[Optional[int]]) -> Dict[date, int]:
|
||
counts: Dict[date, int] = {}
|
||
for ts in timestamps:
|
||
if ts is None:
|
||
continue
|
||
day = datetime.fromtimestamp(ts / 1000).date()
|
||
counts[day] = counts.get(day, 0) + 1
|
||
return counts
|