跑真任务时踩到的:
sqlalchemy.exc.IntegrityError: (1062, "Duplicate entry
'7-7690458980574358513-45' for key 'uq_note_metric'")
根因是我上一版写错了一处作用域:退化路径里那个「遍历已知作品」的循环写在了**目标循环内部**,
所以任务有多个目标时,同一批已知作品会被拉两遍 → 同一件作品在一轮里出现两条记录 →
ingest 给同一件作品写两份本轮快照 → 撞 (task_id, note_id, run_id) 唯一键。
两处都修,各挡一层:
* douyin_fetch:去重集合挪到 collect 的最外层,**跨目标**只算一次;退化时也先查一遍
已知作品是否已刷过。
* ingest:`_ingest_notes` 对「一轮里重复出现的 note_id」免疫。一层在源头、一层在入口,
因为产物里重复并不罕见(多个目标指向同一个人、上游重跑、退化路径),不该靠上游自觉。
测试 +2:多目标时已知作品只刷一次;同一轮里重复的作品只落一份快照(这条会崩在
唯一键上,所以它测的正是运行时的那个崩法)。
151 lines
6.1 KiB
Python
151 lines
6.1 KiB
Python
# -*- coding: utf-8 -*-
|
||
# Copyright (c) 2025 [email protected]
|
||
#
|
||
# This file is part of MediaCrawler project.
|
||
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/douyin_fetch.py
|
||
# GitHub: https://github.com/NanmiCoder
|
||
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
|
||
#
|
||
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
|
||
# 1. 不得用于任何商业用途。
|
||
# 2. 使用时应遵守对应平台的使用条款和robots.txt规则。
|
||
# 3. 不得进行大规模爬取或对平台造成运营干扰。
|
||
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
|
||
# 5. 不得用于任何非法或不当的用途。
|
||
#
|
||
# 详细许可条款请参阅项目根目录下的LICENSE文件。
|
||
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
|
||
|
||
"""抖音采集:直接走 Web 接口,不起爬虫子进程。
|
||
|
||
与 ``media_platform/douyin`` 那条路的分工:
|
||
|
||
* 那边起一个 Playwright 子进程、构造一大串**自相矛盾的浏览器指纹参数**(参数说自己是
|
||
Mac + Chrome 125,UA 说自己是 Linux + Chrome 155),网关回一个 200 + 空 body,
|
||
然后被翻译成 ``Exception("account blocked")`` —— 看起来像账号被封,其实什么都不是。
|
||
* 这边不起子进程、只发必要参数,用浏览器里那份登录态。见 ``douyin_api``。
|
||
|
||
采到的东西写成 ``store/douyin`` 那套 jsonl 形状,所以 **ingest 完全不知道数据是怎么来的**
|
||
—— 重采样、差分、事件、通知、报表全都照旧。
|
||
"""
|
||
|
||
import json
|
||
from datetime import datetime
|
||
from pathlib import Path
|
||
from typing import Any, Dict, Iterable, List, Sequence
|
||
|
||
from tools import utils
|
||
|
||
from . import adapters, douyin_api
|
||
from .models import MODE_CREATOR, MonitorTask
|
||
|
||
# 拉多少条作品。接口单页上限就是 20,要多了也没用。
|
||
DEFAULT_VIDEO_LIMIT = 20
|
||
|
||
|
||
async def collect(
|
||
out_dir: Path,
|
||
*,
|
||
platform: str,
|
||
mode: str,
|
||
limit: int,
|
||
want_comments: bool,
|
||
comment_limit: int,
|
||
targets: Sequence[Any],
|
||
known_aweme_ids: Iterable[str] = (),
|
||
cookie: str = "",
|
||
) -> Dict[str, Any]:
|
||
"""跑一轮抖音采集,把产物写进 ``out_dir`` 下 store 那个目录里。
|
||
|
||
参数是散的、不收 ORM 对象:调用方那边 ``task`` 在会话关掉之后就 detached 了,
|
||
传对象进来迟早会踩到「属性已过期」。
|
||
|
||
**不抛异常**:失败也把(可能为空的)产物落下去,并把原因放进 ``errors`` 交给调用方。
|
||
让调用方去决定这是「一轮正常但没数据」还是「一轮失败」—— 这个判断不该藏在这里。
|
||
"""
|
||
notes: List[Dict[str, Any]] = []
|
||
comments: List[Dict[str, Any]] = []
|
||
errors: List[str] = []
|
||
|
||
# **整个 collect 只去重一次的、跨目标的集合**:退化路径会把「库里已知的全部作品」
|
||
# 在每个目标下都刷一遍,多个目标就会出现同一件作品好几条记录 —— 而一对一快照的
|
||
# 唯一键是 (task_id, note_id, run_id),同一条作品在一轮里出现两次会直接撞键。
|
||
seen_aweme: set = set()
|
||
|
||
for target in targets:
|
||
sec_user_id = target.external_id
|
||
|
||
# 作品列表是主路径;它被那道真校验挡着时,退化到「标题/昵称靠主页接口,作品靠
|
||
# 已知 id 逐条刷新」—— 拿不到新作品,但已知作品的指标还能继续更新。
|
||
try:
|
||
videos = await douyin_api.author_videos(
|
||
sec_user_id, count=limit, cookie=cookie
|
||
)
|
||
except douyin_api.DouyinApiError as exc:
|
||
errors.append(f"拉取博主 {sec_user_id} 的作品列表失败:{exc}")
|
||
videos = []
|
||
for aweme_id in known_aweme_ids:
|
||
if aweme_id in seen_aweme:
|
||
continue
|
||
try:
|
||
videos.append(await douyin_api.video_detail(aweme_id, cookie=cookie))
|
||
except douyin_api.DouyinApiError as detail_exc:
|
||
errors.append(f"刷新作品 {aweme_id} 失败:{detail_exc}")
|
||
|
||
for video in videos:
|
||
aweme_id = video.get("aweme_id")
|
||
if not aweme_id or aweme_id in seen_aweme:
|
||
continue
|
||
seen_aweme.add(aweme_id)
|
||
notes.append(video)
|
||
|
||
if want_comments:
|
||
try:
|
||
comments.extend(
|
||
await douyin_api.video_comments(
|
||
aweme_id, count=comment_limit, cookie=cookie
|
||
)
|
||
)
|
||
except douyin_api.DouyinApiError as exc:
|
||
errors.append(f"拉取作品 {aweme_id} 的评论失败:{exc}")
|
||
|
||
jsonl_dir = _write_artifacts(out_dir, platform, mode, notes, comments)
|
||
return {
|
||
"notes": len(notes),
|
||
"comments": len(comments),
|
||
"errors": errors,
|
||
"jsonl_dir": str(jsonl_dir),
|
||
}
|
||
|
||
|
||
def _write_artifacts(
|
||
out_dir: Path,
|
||
platform: str,
|
||
mode: str,
|
||
notes: List[Dict[str, Any]],
|
||
comments: List[Dict[str, Any]],
|
||
) -> Path:
|
||
"""按爬虫那套目录与文件名写 jsonl。
|
||
|
||
目录名取 ``adapters.artifact_dir``(抖音是 ``douyin``,而不是平台 id ``dy``)——
|
||
和 ingest 找文件用的是同一个来源,两边不会走散。
|
||
"""
|
||
jsonl_dir = out_dir / adapters.artifact_dir(platform) / "jsonl"
|
||
jsonl_dir.mkdir(parents=True, exist_ok=True)
|
||
|
||
kind = "creator" if mode == MODE_CREATOR else "detail"
|
||
date = datetime.now().strftime("%Y-%m-%d")
|
||
|
||
_write_jsonl(jsonl_dir / f"{kind}_contents_{date}.jsonl", notes)
|
||
# 评论文件即使没有评论也建出来:ingest 靠「文件在不在」区分「这一轮没评论」和
|
||
# 「这一轮什么都没抓到」,两种情况的含义完全不同。
|
||
_write_jsonl(jsonl_dir / f"{kind}_comments_{date}.jsonl", comments)
|
||
return jsonl_dir
|
||
|
||
|
||
def _write_jsonl(path: Path, records: List[Dict[str, Any]]) -> None:
|
||
with path.open("w", encoding="utf-8") as handle:
|
||
for record in records:
|
||
handle.write(json.dumps(record, ensure_ascii=False) + "\n")
|
||
utils.logger.info(f"[douyin_fetch] 写出 {len(records)} 条 -> {path.name}")
|