# -*- coding: utf-8 -*- # Copyright (c) 2025 relakkes@gmail.com # # This file is part of MediaCrawler project. # Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/douyin_fetch.py # GitHub: https://github.com/NanmiCoder # Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 # # 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: # 1. 不得用于任何商业用途。 # 2. 使用时应遵守对应平台的使用条款和robots.txt规则。 # 3. 不得进行大规模爬取或对平台造成运营干扰。 # 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 # 5. 不得用于任何非法或不当的用途。 # # 详细许可条款请参阅项目根目录下的LICENSE文件。 # 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 """抖音采集:直接走 Web 接口,不起爬虫子进程。 与 ``media_platform/douyin`` 那条路的分工: * 那边起一个 Playwright 子进程、构造一大串**自相矛盾的浏览器指纹参数**(参数说自己是 Mac + Chrome 125,UA 说自己是 Linux + Chrome 155),网关回一个 200 + 空 body, 然后被翻译成 ``Exception("account blocked")`` —— 看起来像账号被封,其实什么都不是。 * 这边不起子进程、只发必要参数,用浏览器里那份登录态。见 ``douyin_api``。 采到的东西写成 ``store/douyin`` 那套 jsonl 形状,所以 **ingest 完全不知道数据是怎么来的** —— 重采样、差分、事件、通知、报表全都照旧。 """ import json from datetime import datetime from pathlib import Path from typing import Any, Dict, Iterable, List, Sequence from tools import utils from . import adapters, douyin_api from .models import MODE_CREATOR, MonitorTask # 拉多少条作品。接口单页上限就是 20,要多了也没用。 DEFAULT_VIDEO_LIMIT = 20 async def collect( out_dir: Path, *, platform: str, mode: str, limit: int, want_comments: bool, comment_limit: int, targets: Sequence[Any], known_aweme_ids: Iterable[str] = (), cookie: str = "", ) -> Dict[str, Any]: """跑一轮抖音采集,把产物写进 ``out_dir`` 下 store 那个目录里。 参数是散的、不收 ORM 对象:调用方那边 ``task`` 在会话关掉之后就 detached 了, 传对象进来迟早会踩到「属性已过期」。 **不抛异常**:失败也把(可能为空的)产物落下去,并把原因放进 ``errors`` 交给调用方。 让调用方去决定这是「一轮正常但没数据」还是「一轮失败」—— 这个判断不该藏在这里。 """ notes: List[Dict[str, Any]] = [] comments: List[Dict[str, Any]] = [] errors: List[str] = [] # **整个 collect 只去重一次的、跨目标的集合**:退化路径会把「库里已知的全部作品」 # 在每个目标下都刷一遍,多个目标就会出现同一件作品好几条记录 —— 而一对一快照的 # 唯一键是 (task_id, note_id, run_id),同一条作品在一轮里出现两次会直接撞键。 seen_aweme: set = set() for target in targets: sec_user_id = target.external_id # 作品列表是主路径;它被那道真校验挡着时,退化到「标题/昵称靠主页接口,作品靠 # 已知 id 逐条刷新」—— 拿不到新作品,但已知作品的指标还能继续更新。 try: videos = await douyin_api.author_videos( sec_user_id, count=limit, cookie=cookie ) except douyin_api.DouyinApiError as exc: errors.append(f"拉取博主 {sec_user_id} 的作品列表失败:{exc}") videos = [] for aweme_id in known_aweme_ids: if aweme_id in seen_aweme: continue try: videos.append(await douyin_api.video_detail(aweme_id, cookie=cookie)) except douyin_api.DouyinApiError as detail_exc: errors.append(f"刷新作品 {aweme_id} 失败:{detail_exc}") for video in videos: aweme_id = video.get("aweme_id") if not aweme_id or aweme_id in seen_aweme: continue seen_aweme.add(aweme_id) notes.append(video) if want_comments: try: comments.extend( await douyin_api.video_comments( aweme_id, count=comment_limit, cookie=cookie ) ) except douyin_api.DouyinApiError as exc: errors.append(f"拉取作品 {aweme_id} 的评论失败:{exc}") jsonl_dir = _write_artifacts(out_dir, platform, mode, notes, comments) return { "notes": len(notes), "comments": len(comments), "errors": errors, "jsonl_dir": str(jsonl_dir), } def _write_artifacts( out_dir: Path, platform: str, mode: str, notes: List[Dict[str, Any]], comments: List[Dict[str, Any]], ) -> Path: """按爬虫那套目录与文件名写 jsonl。 目录名取 ``adapters.artifact_dir``(抖音是 ``douyin``,而不是平台 id ``dy``)—— 和 ingest 找文件用的是同一个来源,两边不会走散。 """ jsonl_dir = out_dir / adapters.artifact_dir(platform) / "jsonl" jsonl_dir.mkdir(parents=True, exist_ok=True) kind = "creator" if mode == MODE_CREATOR else "detail" date = datetime.now().strftime("%Y-%m-%d") _write_jsonl(jsonl_dir / f"{kind}_contents_{date}.jsonl", notes) # 评论文件即使没有评论也建出来:ingest 靠「文件在不在」区分「这一轮没评论」和 # 「这一轮什么都没抓到」,两种情况的含义完全不同。 _write_jsonl(jsonl_dir / f"{kind}_comments_{date}.jsonl", comments) return jsonl_dir def _write_jsonl(path: Path, records: List[Dict[str, Any]]) -> None: with path.open("w", encoding="utf-8") as handle: for record in records: handle.write(json.dumps(record, ensure_ascii=False) + "\n") utils.logger.info(f"[douyin_fetch] 写出 {len(records)} 条 -> {path.name}")