# -*- coding: utf-8 -*- # Copyright (c) 2025 relakkes@gmail.com # # This file is part of MediaCrawler project. # Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/platforms.py # GitHub: https://github.com/NanmiCoder # Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 # # 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: # 1. 不得用于任何商业用途。 # 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 # 3. 不得进行大规模爬取或对平台造成运营干扰。 # 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 # 5. 不得用于任何非法或不当的用途。 # # 详细许可条款请参阅项目根目录下的LICENSE文件。 # 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 """Platform capability matrix. The single source of truth for what each platform can do. The UI renders its platform switcher and metric columns from this, and the API validates against it. Two distinct things are recorded here, and conflating them would be misleading: * ``crawler_modes`` / ``metrics`` / ``comment_levels`` / ``media`` describe what the upstream crawler module actually supports. These were read out of the platform modules, not assumed -- all seven implement search/detail/creator; the real differences are in which interaction metrics they capture. * ``monitor_wired`` says whether the *monitoring layer* has been hooked up. It covers Xiaohongshu and Douyin. The parts where those two differ -- which directory the crawler writes into, what the jsonl fields are called, what a target URL looks like -- live in ``adapters.py``; the rest of the layer is platform-neutral. A platform can therefore be fully crawlable by upstream and still not usable for monitoring, which is exactly the state of the other five today. """ from typing import Any, Dict, List, Optional PLATFORM_XHS = "xhs" PLATFORM_LABELS = { "xhs": "小红书", "dy": "抖音", "ks": "快手", "bili": "B站", "wb": "微博", "tieba": "贴吧", "zhihu": "知乎", } # Interaction metrics each platform's store actually persists. Xiaohongshu has no # play count or danmaku; Bilibili has both and the widest set; Kuaishou carries # no comment/share/collect at all; Tieba stores only reply counts. PLATFORM_CAPABILITIES: Dict[str, Dict[str, Any]] = { "xhs": { "crawler_modes": ["search", "detail", "creator"], "metrics": ["liked_count", "comment_count", "collected_count", "share_count"], "comment_levels": 2, "media": True, "monitor_wired": True, "target_hints": { "creator": "https://www.xiaohongshu.com/user/profile/5f58bd990000000001003753", "note": "https://www.xiaohongshu.com/explore/6aa3d827000000002802c5c8?xsec_token=...", "creator_label": "博主主页", "note_label": "笔记", # 只有小红书的链接带会过期的 xsec_token。抖音的链接不带令牌,永久有效, # 那句「建议只填纯 ID」的劝告对它没有意义。 "token_expires": True, }, }, "dy": { "crawler_modes": ["search", "detail", "creator"], "metrics": ["liked_count", "comment_count", "collected_count", "share_count"], "comment_levels": 2, "media": True, "monitor_wired": True, # 用户可见的示例链接(前端的目标输入框用它做 placeholder)。放这里是因为 # 它属于「这个平台长什么样」的能力描述;真正干活的管子(正则、目录名、 # 字段别名)在 adapters.py。 "target_hints": { "creator": "https://www.douyin.com/user/MS4wLjABAAAATJPY7LAlaa5X-c8uNdWkvz0jUGgpw4eeXIwu_8BhvqE", "note": "https://www.douyin.com/video/7525082444551310602", "creator_label": "博主主页", "note_label": "作品", }, }, "ks": { "crawler_modes": ["search", "detail", "creator"], # No comment/share/collect in the Kuaishou store; sub-comments are stored # flat with no parent link and carry no like count. "metrics": ["liked_count", "view_count"], "comment_levels": 1, "media": True, "monitor_wired": False, }, "bili": { "crawler_modes": ["search", "detail", "creator"], "metrics": [ "liked_count", "video_play_count", "video_danmaku", "comment_count", "video_favorite_count", "video_coin_count", "video_share_count", ], "comment_levels": 2, "media": True, "monitor_wired": False, }, "wb": { "crawler_modes": ["search", "detail", "creator"], # Weibo has no collect count, and its comment count field is named # differently in the model. "metrics": ["liked_count", "comments_count", "shared_count"], "comment_levels": 2, "media": True, "monitor_wired": False, }, "tieba": { "crawler_modes": ["search", "detail", "creator"], "metrics": ["total_replay_num", "total_replay_page"], "comment_levels": 2, "media": False, "monitor_wired": False, }, "zhihu": { "crawler_modes": ["search", "detail", "creator"], "metrics": ["voteup_count", "comment_count"], "comment_levels": 2, "media": False, "monitor_wired": False, }, } METRIC_LABELS = { "liked_count": "点赞", "comment_count": "评论", "collected_count": "收藏", "share_count": "分享", "view_count": "播放", "video_play_count": "播放", "video_danmaku": "弹幕", "video_favorite_count": "收藏", "video_coin_count": "投币", "video_share_count": "分享", "comments_count": "评论", "shared_count": "转发", "total_replay_num": "回复数", "total_replay_page": "回复页数", "voteup_count": "赞同", } # Monitoring modes, mapped to the CLI crawler types upstream understands. MONITOR_MODE_CREATOR = "creator" MONITOR_MODE_NOTE = "note" CLI_TYPE_FOR_MODE = { MONITOR_MODE_CREATOR: "creator", MONITOR_MODE_NOTE: "detail", } class UnsupportedPlatformError(ValueError): """Raised for an unknown platform, or one the monitor layer cannot run.""" def all_platforms() -> List[str]: return list(PLATFORM_CAPABILITIES) def is_known(platform: str) -> bool: return platform in PLATFORM_CAPABILITIES def is_monitor_wired(platform: str) -> bool: return bool(PLATFORM_CAPABILITIES.get(platform, {}).get("monitor_wired")) def describe(platform: str) -> Optional[Dict[str, Any]]: capability = PLATFORM_CAPABILITIES.get(platform) if capability is None: return None return { "value": platform, "label": PLATFORM_LABELS.get(platform, platform), **capability, "metric_labels": { metric: METRIC_LABELS.get(metric, metric) for metric in capability["metrics"] }, } def describe_all() -> List[Dict[str, Any]]: return [describe(platform) for platform in all_platforms()] def ensure_runnable(platform: str) -> None: """Validate a platform for a monitoring task. An unwired platform is rejected outright rather than accepted and left to silently produce nothing -- the same silent-failure shape that made a valid creator look like an expired login earlier. """ if not is_known(platform): raise UnsupportedPlatformError( f"未知平台:{platform}(支持:{', '.join(all_platforms())})" ) if not is_monitor_wired(platform): label = PLATFORM_LABELS.get(platform, platform) raise UnsupportedPlatformError( f"{label}的爬虫已支持,但监控层尚未接通,暂时无法创建监控任务。" )