# -*- coding: utf-8 -*- # Copyright (c) 2025 relakkes@gmail.com # # This file is part of MediaCrawler project. # Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/report.py # GitHub: https://github.com/NanmiCoder # Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 # # 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: # 1. 不得用于任何商业用途。 # 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 # 3. 不得进行大规模爬取或对平台造成运营干扰。 # 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 # 5. 不得用于任何非法或不当的用途。 # # 详细许可条款请参阅项目根目录下的LICENSE文件。 # 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 """Cross-task reporting: what grew, and what is new, over a date range. Two families of numbers that answer different questions and are therefore kept as separate columns: * **互动增量** — Σ(current − previous) across the selected notes. "How many likes did this set of notes gain?" * **新增内容** — count of newly discovered notes and comments. "How much new material showed up?" The per-day interaction delta is defined as *last value on the day* minus *last value before the day* (0 when the note was first seen on that day). That keeps growth from a note's first observation counted once, rather than smeared across every later day. Aggregation runs in Python over the snapshots rather than as one large SQL query: the per-note-per-day baseline lookup is a windowed operation that SQLite expresses awkwardly, and the row counts here are small enough that clarity is worth more than the query planner. """ from bisect import bisect_right from datetime import date, datetime, time, timedelta from typing import Any, Dict, Iterable, List, Optional, Sequence from sqlalchemy import select from sqlalchemy.ext.asyncio import AsyncSession from .models import MonitorComment, MonitorNote, MonitorNoteMetric METRIC_FIELDS = ("liked_count", "comment_count", "collected_count", "share_count") METRIC_LABELS = { "liked_count": "点赞", "comment_count": "评论", "collected_count": "收藏", "share_count": "分享", } def day_bounds(day: date) -> tuple[int, int]: """Inclusive epoch-millisecond bounds for a local calendar day.""" start = datetime.combine(day, time.min) end = datetime.combine(day, time.max) return int(start.timestamp() * 1000), int(end.timestamp() * 1000) def iter_days(start: date, end: date) -> List[date]: days = [] cursor = start while cursor <= end: days.append(cursor) cursor += timedelta(days=1) return days def compute_daily_rows( series_by_note: Dict[str, List[tuple[int, Dict[str, Optional[int]]]]], notes_per_day: Dict[date, int], comments_per_day: Dict[date, int], days: Sequence[date], ) -> List[Dict[str, Any]]: """Pure aggregation. ``series_by_note`` must be sorted by timestamp ascending.""" prepared = {note_id: ([ts for ts, _ in points], points) for note_id, points in series_by_note.items()} rows: List[Dict[str, Any]] = [] for day in days: day_start, day_end = day_bounds(day) totals = {field: 0 for field in METRIC_FIELDS} # Records *which* metric could not be compared, not just that something # could not. A blanket flag loses all value the moment one permanently # unparseable field makes every row "incomplete". partial_metrics: set[str] = set() for times, points in prepared.values(): end_index = bisect_right(times, day_end) - 1 if end_index < 0: # Not yet tracked on this day. continue end_values = points[end_index][1] start_index = bisect_right(times, day_start - 1) - 1 # No earlier snapshot means the note first appeared in this window, # so it starts from zero -- all of its count is genuinely new. start_values = ( points[start_index][1] if start_index >= 0 else {f: 0 for f in METRIC_FIELDS} ) for field in METRIC_FIELDS: end_value, start_value = end_values.get(field), start_values.get(field) if end_value is None or start_value is None: # An unparseable count on either side makes the delta unknown; # skipping beats reporting a fabricated number. partial_metrics.add(field) continue totals[field] += end_value - start_value row: Dict[str, Any] = { "date": day.isoformat(), "new_notes": notes_per_day.get(day, 0), "new_comments": comments_per_day.get(day, 0), "partial_metrics": sorted(partial_metrics), } row.update({f"{field}_delta": value for field, value in totals.items()}) rows.append(row) return rows async def build_report( session: AsyncSession, task_ids: Optional[Iterable[int]], start_day: date, end_day: date, ) -> Dict[str, Any]: """Daily rows plus totals for the selected tasks over the given date range.""" start_ms, _ = day_bounds(start_day) _, end_ms = day_bounds(end_day) scope = list(task_ids) if task_ids else None days = iter_days(start_day, end_day) # Fetch every snapshot up to the range end: the delta on the first day needs # the last value from *before* the range, so a lower bound would be wrong. metric_stmt = select(MonitorNoteMetric).where(MonitorNoteMetric.captured_at <= end_ms) if scope is not None: metric_stmt = metric_stmt.where(MonitorNoteMetric.task_id.in_(scope)) metric_stmt = metric_stmt.order_by(MonitorNoteMetric.note_id, MonitorNoteMetric.run_id) series_by_note: Dict[str, List[tuple[int, Dict[str, Optional[int]]]]] = {} included_note_ids: set[str] = set() for snapshot in (await session.scalars(metric_stmt)).all(): included_note_ids.add(snapshot.note_id) series_by_note.setdefault(snapshot.note_id, []).append( ( snapshot.captured_at, {field: getattr(snapshot, field) for field in METRIC_FIELDS}, ) ) note_stmt = select(MonitorNote.first_seen_at).where( MonitorNote.first_seen_at >= start_ms, MonitorNote.first_seen_at <= end_ms ) if scope is not None: note_stmt = note_stmt.where(MonitorNote.task_id.in_(scope)) comment_stmt = select(MonitorComment.first_seen_at).where( MonitorComment.first_seen_at >= start_ms, MonitorComment.first_seen_at <= end_ms ) if scope is not None: comment_stmt = comment_stmt.where(MonitorComment.task_id.in_(scope)) notes_per_day = _count_by_day((await session.scalars(note_stmt)).all()) comments_per_day = _count_by_day((await session.scalars(comment_stmt)).all()) rows = compute_daily_rows(series_by_note, notes_per_day, comments_per_day, days) totals = { "new_notes": sum(row["new_notes"] for row in rows), "new_comments": sum(row["new_comments"] for row in rows), } for field in METRIC_FIELDS: totals[f"{field}_delta"] = sum(row[f"{field}_delta"] for row in rows) return { "start_date": start_day.isoformat(), "end_date": end_day.isoformat(), "task_ids": scope, "rows": rows, "totals": totals, "note_count": len(included_note_ids), "has_partial_data": any(row["partial_metrics"] for row in rows), "partial_metrics": sorted({field for row in rows for field in row["partial_metrics"]}), "metric_labels": METRIC_LABELS, } def _count_by_day(timestamps: Iterable[Optional[int]]) -> Dict[date, int]: counts: Dict[date, int] = {} for ts in timestamps: if ts is None: continue day = datetime.fromtimestamp(ts / 1000).date() counts[day] = counts.get(day, 0) + 1 return counts