diff --git a/.gitignore b/.gitignore index 87bf8fc..4905579 100644 --- a/.gitignore +++ b/.gitignore @@ -183,4 +183,9 @@ agent_zone debug_tools database/*.db -.omx/ \ No newline at end of file +.omx/ + +# 别人放在这儿的参考项目(mac-agent-os)。它是独立仓库、21MB,不属于本项目 —— +# 一旦被 `git add -A` 扫进来就是永久留在历史里(踩过一次:1601 个文件里 1429 个是它)。 +# 要读它就直接读磁盘上的目录,别提交。 +mac-agent-os-main/ \ No newline at end of file diff --git a/api/monitor/douyin_api.py b/api/monitor/douyin_api.py new file mode 100644 index 0000000..5260c28 --- /dev/null +++ b/api/monitor/douyin_api.py @@ -0,0 +1,429 @@ +# -*- coding: utf-8 -*- +# Copyright (c) 2025 relakkes@gmail.com +# +# This file is part of MediaCrawler project. +# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/douyin_api.py +# GitHub: https://github.com/NanmiCoder +# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 +# +# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: +# 1. 不得用于任何商业用途。 +# 2. 使用时应遵守对应平台的使用条款和robots.txt规则。 +# 3. 不得进行大规模爬取或对平台造成运营干扰。 +# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 +# 5. 不得用于任何非法或不当的用途。 +# +# 详细许可条款请参阅项目根目录下的LICENSE文件。 +# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 + +"""抖音 Web 接口客户端 —— 直接发 HTTP,不起爬虫子进程。 + +**为什么另起一套。** 爬虫那条路(``media_platform/douyin``)会构造一大串浏览器指纹 +参数:``browser_platform=MacIntel``、``os_name=Mac OS``、``browser_version=125.0.0.0``…… +而 ``User-Agent`` 是从页面现读的(在服务器上是 Linux + Chrome 155)。参数说自己是 Mac, +UA 说自己是 Linux —— 抖音网关对这种自相矛盾的请求的处理方式是:**不报错、不给原因, +回一个 200 + 空 body**。爬虫那边把它翻译成 ``Exception("account blocked")``,看起来像 +账号被封,其实什么都不是。 + +这份客户端只发必要参数(``device_platform`` / ``aid`` 那两三个),走浏览器自己也在用的 +那条调用路径。它的做法来自 mac-agent-os 项目的 ``mediacrawler_adapter.py``,实测可用。 + +两个关键点: + +* **cookie 走 CDP 现读。** Chrome 把 cookie 值加密存在 SQLite 里,只有 CDP 拿得到 + 解密后的值;而且浏览器里那份比库里存的旧快照新 —— 站点会自己轮换会话。 +* **产物形状照抄 store。** ``aweme_id`` / ``aweme_url`` / ``cover_url`` / ``aweme_type`` / + ``create_time``(**秒**,由 adapters 换算成毫秒)…… 这样 ingest 那条链路一个字都不用改。 +""" + +import asyncio +import os +import time +from dataclasses import dataclass +from typing import Any, Dict, List, Optional, Tuple + +import config +import httpx +from tools import utils +from tools.user_hash import anonymize_user_id + +# 请求头。**要像一个浏览器**,而且必须是**同一个浏览器**:见 BrowserIdentity。 +_BASE_HEADERS = { + "Accept": "application/json, text/plain, */*", + "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8", + "Referer": "https://www.douyin.com/", + "Origin": "https://www.douyin.com", +} + +# 网关的业务前置校验头。缺了它,抖音边缘网关的 ArgusSecurityPlugin 会直接回 +# 403 并写明 "Blocked by ArgusSecurityPlugin Uifid Not Found" —— 难得一次它会说原因。 +# 当前网关并不校验这个头的**值**,填什么都行;一旦升级到真校验,就得改成让页面里的 +# SDK 自己生成(见 media_platform/douyin/client.py 里同一条注释)。 +ARGUS_HEADER_VALUE = "1" + +API_ORIGIN = "https://www.douyin.com" +PROFILE_PATH = "/aweme/v1/web/user/profile/other/" +POSTS_PATH = "/aweme/v1/web/aweme/post/" +DETAIL_PATH = "/aweme/v1/web/aweme/detail/" +COMMENT_PATH = "/aweme/v1/web/comment/list/" + +# 一次请求的超时。抖音这两个接口正常都在一秒内返回。 +REQUEST_TIMEOUT_SECONDS = 20.0 +# 单页最多要多少条。接口自己有上限,要多了也没用。 +MAX_PAGE_SIZE = 20 + + +class DouyinApiError(RuntimeError): + """请求失败,或登录态不可用。""" + + +def _cdp_url() -> str: + """浏览器 DevTools 端点。与扫码登录那边共用同一个开关。""" + return os.getenv("MC_CDP_URL") or f"http://127.0.0.1:{config.CDP_DEBUG_PORT}" + + +@dataclass +class BrowserIdentity: + """一个请求要像浏览器所需要的全部身份信息,**且必须来自同一个浏览器**。 + + 只拿 cookie 是不够的。UA 声称自己是 Chrome 155、却不带 Chrome 155 该有的 + ``sec-ch-ua``,网关一眼就能看出这不是浏览器 —— 它的回应是 **200 + 空 body**: + 不报错、不给原因,只看得到「抓到 0 条」。所以这三样必须成套地从同一处取。 + """ + + cookie: str + user_agent: str + client_hints: Dict[str, str] + + def headers(self) -> Dict[str, str]: + headers = { + "User-Agent": self.user_agent, + **self.client_hints, + **_BASE_HEADERS, + "x-tt-argus": ARGUS_HEADER_VALUE, + "Cookie": self.cookie, + } + # uifid 是设备标识,网关要它;cookie 里没有就不带(送空值反而更像异常请求)。 + uifid = _cookie_value(self.cookie, "UIFID") or _cookie_value( + self.cookie, "UIFID_TEMP" + ) + if uifid: + headers["uifid"] = uifid + return headers + + +# 身份信息的短时缓存:一次采集要发好几个请求,没必要每次都连一遍 CDP。 +_IDENTITY_TTL_SECONDS = 120.0 +_identity_cache: Optional[Tuple[float, BrowserIdentity]] = None + + +async def _read_browser() -> Optional[BrowserIdentity]: + """连上 CDP 浏览器,一次取齐 cookie、UA、client hints。 + + 读不到返回 None(浏览器没开/没登录),由调用方决定怎么报 —— 不抛异常。 + """ + from playwright.async_api import async_playwright + + from media_platform.douyin.help import client_hint_headers + + playwright = None + try: + playwright = await async_playwright().start() + browser = await playwright.chromium.connect_over_cdp(_cdp_url(), timeout=15000) + if not browser.contexts: + return None + # contexts[0] 是真实 profile。**不要 new_context()** —— 那是无痕式的,读不到登录态。 + context = browser.contexts[0] + cookies = await context.cookies() + + # UA 和 hints 要从页面里问 —— 它们是浏览器自己的事实,写死迟早对不上。 + page = context.pages[0] if context.pages else await context.new_page() + user_agent = await page.evaluate("() => navigator.userAgent") + hints = client_hint_headers( + await page.evaluate("() => navigator.userAgentData || null") + ) + except Exception as exc: + utils.logger.warning(f"[douyin_api] 读浏览器身份失败:{exc}") + return None + finally: + if playwright is not None: + # 只断开连接。**绝不能 browser.close()** —— 对这个 CDP 连接而言那会关掉 + # 操作者自己的浏览器。 + try: + await playwright.stop() + except Exception: + pass + + douyin_cookies = { + cookie["name"]: cookie["value"] + for cookie in cookies + if "douyin" in cookie.get("domain", "") or "amemv" in cookie.get("domain", "") + } + return BrowserIdentity( + cookie=_cookie_from_dict(douyin_cookies), + user_agent=user_agent or "", + client_hints=hints, + ) + + +async def browser_identity(cookie: str = "", force: bool = False) -> BrowserIdentity: + """拿到一份可用的身份:**优先浏览器里那份**,其次退回传进来的 cookie(库里存的)。 + + 优先浏览器的原因:站点会自己轮换会话,库里存的是粘贴那一刻的快照,浏览器里那份才是 + 当前有效的;而 UA/hints 更是只有浏览器自己知道。 + """ + global _identity_cache + + now = time.monotonic() + if not force and _identity_cache is not None: + cached_at, cached = _identity_cache + if now - cached_at < _IDENTITY_TTL_SECONDS: + return cached + + identity = await _read_browser() + if identity is None or not _has_session(identity.cookie): + # 浏览器里没有可用会话,退回调用方给的那份。UA/hints 编不出来就不编 —— + # 一组和 UA 对不上的 hints 比没有更糟。 + identity = BrowserIdentity( + cookie=_cookie_header(cookie), user_agent="", client_hints={} + ) + _identity_cache = (now, identity) + return identity + + +def forget_identity() -> None: + """丢掉缓存的身份。cookie 变了、或测试之间要隔离时调用。""" + global _identity_cache + _identity_cache = None + + +def _cookie_header(cookie: str) -> str: + """把 ``a=1; b=2`` 形式的 cookie 串规整成请求头用的形状。""" + pairs = [] + for part in (cookie or "").split(";"): + if "=" in part: + name, _, value = part.partition("=") + name = name.strip() + if name: + pairs.append(f"{name}={value.strip()}") + return "; ".join(pairs) + + +def _cookie_from_dict(cookies: Dict[str, str]) -> str: + return "; ".join(f"{name}={value}" for name, value in cookies.items()) + + +def _cookie_value(cookie: str, name: str) -> str: + """从一个 cookie 串里取某个键的值。""" + for part in (cookie or "").split(";"): + key, _, value = part.partition("=") + if key.strip() == name: + return value.strip() + return "" + + +async def _get( + path: str, params: Dict[str, Any], identity: BrowserIdentity +) -> Dict[str, Any]: + """发一个 GET,返回 JSON。 + + 只带调用方给的参数 —— **不要往里加 webid / msToken / browser_version 那一堆**, + 那正是爬虫那条路失败的原因。 + """ + url = f"{API_ORIGIN}{path}" + async with httpx.AsyncClient(timeout=REQUEST_TIMEOUT_SECONDS) as client: + response = await client.get( + url, + params=params, + headers=identity.headers(), + ) + + if response.status_code != 200: + raise DouyinApiError(f"HTTP {response.status_code}:{response.text[:120]}") + + # 「200 + 空 body」是抖音网关拒绝请求时的典型回应(见模块说明)。必须当成错误报出来, + # 否则会一路往下变成「这个博主没作品」。 + if not response.text.strip(): + raise DouyinApiError( + "接口返回了空内容 —— 通常是登录态失效,或请求被网关判成了非浏览器" + ) + + try: + return response.json() + except ValueError as exc: + raise DouyinApiError(f"返回的不是 JSON:{response.text[:120]}") from exc + + +def _as_int(value: Any) -> int: + try: + return int(value) + except (TypeError, ValueError): + return 0 + + +def normalize_aweme(aweme: Dict[str, Any]) -> Dict[str, Any]: + """把接口返回的一条作品,翻译成 store 落盘的那套键名。 + + 键名必须和 ``store/douyin`` 一致 —— 跨过这一层之后,ingest 就不知道数据是从爬虫 + 来的还是从接口来的。 + """ + author = aweme.get("author") or {} + statistics = aweme.get("statistics") or {} + aweme_id = str(aweme.get("aweme_id") or "") + cover = ((aweme.get("video") or {}).get("cover") or {}).get("url_list") or [""] + uid = str(author.get("uid") or "") + nickname = author.get("nickname") or "" + + return { + "aweme_id": aweme_id, + "aweme_type": str(aweme.get("aweme_type") or ""), + # store 那边 title 取的是 desc。 + "title": aweme.get("desc") or "", + "desc": aweme.get("desc") or "", + # **秒**。adapters.time_scale 会把它换成毫秒,和 store 写出来的形态一致。 + "create_time": _as_int(aweme.get("create_time")), + "creator_hash": anonymize_user_id(uid or author.get("sec_uid") or ""), + "nickname": nickname, + "liked_count": str(_as_int(statistics.get("digg_count"))), + "comment_count": str(_as_int(statistics.get("comment_count"))), + "collected_count": str(_as_int(statistics.get("collect_count"))), + "share_count": str(_as_int(statistics.get("share_count"))), + "aweme_url": f"https://www.douyin.com/video/{aweme_id}", + "cover_url": cover[0] if cover else "", + "source_keyword": "", + } + + +async def author_videos( + sec_user_id: str, count: int = MAX_PAGE_SIZE, *, cookie: str = "" +) -> List[Dict[str, Any]]: + """某个博主最新发布的作品(按发布时间倒序),已翻译成 store 的键名。 + + 用 ``sec_user_id`` 而不是数字 uid:监控任务里存的就是主页链接里的那段 sec_uid, + 而且这个接口两种都收(爬虫那边用的也是 sec_user_id)。 + """ + identity = await browser_identity(cookie) + if not _has_session(identity.cookie): + raise DouyinApiError("抖音登录态不可用:浏览器里没有会话,库里的 cookie 也没有") + + payload = await _get( + POSTS_PATH, + { + "sec_user_id": sec_user_id, + "count": max(1, min(count, MAX_PAGE_SIZE)), + "max_cursor": 0, + "device_platform": "webapp", + "aid": 6383, + }, + identity, + ) + + awemes = payload.get("aweme_list") or [] + if not awemes and payload.get("status_code") not in (0, None): + raise DouyinApiError( + f"接口拒绝了请求(status_code={payload.get('status_code')})" + ) + return [normalize_aweme(aweme) for aweme in awemes] + + +async def author_profile(sec_user_id: str, *, cookie: str = "") -> Dict[str, Any]: + """博主主页指标:昵称 / 粉丝数 / 总获赞 / 作品数。""" + identity = await browser_identity(cookie) + if not _has_session(identity.cookie): + raise DouyinApiError("抖音登录态不可用:浏览器里没有会话,库里的 cookie 也没有") + + payload = await _get( + PROFILE_PATH, + {"sec_user_id": sec_user_id, "device_platform": "webapp", "aid": 6383}, + identity, + ) + user = payload.get("user") or {} + if not user: + raise DouyinApiError( + f"接口没返回用户数据(status_code={payload.get('status_code')})" + ) + return { + "nickname": user.get("nickname") or "", + "unique_id": user.get("unique_id") or "", + "fans": _as_int(user.get("follower_count")), + "total_favorited": _as_int(user.get("total_favorited")), + "works": _as_int(user.get("aweme_count")), + "following": _as_int(user.get("following_count")), + } + + +async def video_comments( + aweme_id: str, count: int = 20, *, cookie: str = "" +) -> List[Dict[str, Any]]: + """一条作品的评论,翻译成 store 的评论键名。 + + 刻意不带 ``sub_comment_count`` / ``parent_comment_id`` 的猜测值 —— 接口给了就用, + 没给就留空,不编。 + """ + identity = await browser_identity(cookie) + if not _has_session(identity.cookie): + raise DouyinApiError("抖音登录态不可用") + + payload = await _get( + COMMENT_PATH, + { + "aweme_id": aweme_id, + "count": max(1, min(count, MAX_PAGE_SIZE)), + "cursor": 0, + "device_platform": "webapp", + "aid": 6383, + }, + identity, + ) + + records = [] + for comment in payload.get("comments") or []: + user = comment.get("user") or {} + records.append( + { + "comment_id": str(comment.get("cid") or ""), + "aweme_id": aweme_id, + "content": comment.get("text") or "", + "nickname": user.get("nickname") or "", + "creator_hash": anonymize_user_id( + str(user.get("uid") or user.get("sec_uid") or "") + ), + # 同为秒;adapters 会换算。 + "create_time": _as_int(comment.get("create_time")), + "like_count": str(_as_int(comment.get("digg_count"))), + "sub_comment_count": str(_as_int(comment.get("reply_comment_total"))), + # 顶层评论在抖音里是 "0";adapters.parent_comment_id 会归一成空串。 + "parent_comment_id": str(comment.get("reply_id") or "0"), + } + ) + return records + + +def _has_session(cookie: str) -> bool: + return "sessionid=" in (cookie or "") + + +async def check_login(cookie: str = "") -> Dict[str, Any]: + """浏览器/库里现在有没有可用的抖音登录态。给设置页用。""" + identity = await browser_identity(cookie) + if _has_session(identity.cookie): + source = "browser" if identity.user_agent else "stored" + return {"ok": True, "source": source, "cookie_length": len(identity.cookie)} + return {"ok": False, "source": "", "cookie_length": 0} + + +async def main() -> None: # pragma: no cover - 手工排查用 + """``python -m api.monitor.douyin_api ``""" + import sys + + if len(sys.argv) < 2: + print(await check_login()) + return + sec = sys.argv[1] + print(await author_profile(sec)) + for record in await author_videos(sec, count=5): + print(record["create_time"], record["title"][:30], record["liked_count"]) + + +if __name__ == "__main__": # pragma: no cover + asyncio.run(main()) diff --git a/tests/test_douyin_api.py b/tests/test_douyin_api.py new file mode 100644 index 0000000..3feaa50 --- /dev/null +++ b/tests/test_douyin_api.py @@ -0,0 +1,181 @@ +# -*- coding: utf-8 -*- +"""抖音 Web 接口客户端 —— 纯逻辑部分(不发网络请求、不连浏览器)。 + +发请求那半边只能在真环境里验(要 CDP 浏览器 + 登录态),所以这里钉住的是那些 +「错了会一路错到入库」的地方:请求头的成套性、cookie 解析、以及产物键名。 +""" + +from tools.user_hash import anonymize_user_id + +import asyncio + +import httpx +import pytest + +from api.monitor import douyin_api + + +class TestCookieParsing: + def test_cookie_header_is_normalised(self): + assert douyin_api._cookie_header(" a=1 ; b = 2 ;; c=3 ") == "a=1; b=2; c=3" + + def test_cookie_value_lookup(self): + assert douyin_api._cookie_value("a=1; UIFID=xyz; b=2", "UIFID") == "xyz" + assert douyin_api._cookie_value("a=1", "UIFID") == "" + assert douyin_api._cookie_value("", "UIFID") == "" + + def test_session_detection(self): + assert douyin_api._has_session("a=1; sessionid=abc") is True + assert douyin_api._has_session("a=1; sessionid_ss=abc") is False + assert douyin_api._has_session("") is False + + +class TestRequestHeaders: + """请求头必须**成套**,而且成套地来自同一个浏览器。 + + 实测:只有 UA + client hints + Cookie 时,主页接口回 200 但只有 121 字节(空壳); + 补上 Accept / Accept-Language / Referer 才变成 7074 字节的真数据。 + """ + + def test_the_full_set_is_sent(self): + identity = douyin_api.BrowserIdentity( + cookie="sessionid=s; UIFID=u1", + user_agent="UA-of-this-browser", + client_hints={"sec-ch-ua": '"Chrome";v="155"'}, + ) + + headers = identity.headers() + + assert headers["User-Agent"] == "UA-of-this-browser" + assert headers["sec-ch-ua"] == '"Chrome";v="155"' + assert headers["Accept"], "Accept 系列是主页接口能不能返回真数据的必要条件" + assert headers["Accept-Language"] + assert headers["Referer"] == "https://www.douyin.com/" + assert headers["x-tt-argus"] == douyin_api.ARGUS_HEADER_VALUE + assert headers["uifid"] == "u1" + assert headers["Cookie"] == "sessionid=s; UIFID=u1" + + def test_uifid_is_omitted_when_absent(self): + """cookie 里没有 uifid 就别带 —— 送个空值反而更像异常请求。""" + identity = douyin_api.BrowserIdentity( + cookie="sessionid=s", user_agent="UA", client_hints={} + ) + + assert "uifid" not in identity.headers() + + def test_uifid_temp_is_used_as_a_fallback(self): + identity = douyin_api.BrowserIdentity( + cookie="sessionid=s; UIFID_TEMP=temp-1", user_agent="UA", client_hints={} + ) + + assert identity.headers()["uifid"] == "temp-1" + + +class TestNormalizeAweme: + def test_keys_match_what_the_store_writes(self): + """键名必须和 store/douyin 一模一样,否则 ingest 一条都读不到。""" + record = douyin_api.normalize_aweme( + { + "aweme_id": 7690458980574358513, + "desc": "中秋哪儿都堵", + "create_time": 1790574515, + "author": {"uid": "776719710825195", "nickname": "AA建材王总"}, + "statistics": { + "digg_count": 3, + "comment_count": 1, + "collect_count": 2, + "share_count": 0, + }, + "video": {"cover": {"url_list": ["https://img/cover.jpg"]}}, + } + ) + + assert record["aweme_id"] == "7690458980574358513" + assert record["title"] == "中秋哪儿都堵" + assert record["nickname"] == "AA建材王总" + assert record["cover_url"] == "https://img/cover.jpg" + assert ( + record["aweme_url"] + == "https://www.douyin.com/video/7690458980574358513" + ) + # 与 store 一致:creator_hash 是 uid 的匿名哈希。 + assert record["creator_hash"] == anonymize_user_id("776719710825195") + # **秒**。adapters 的 time_scale=1000 会把它换成毫秒 —— 这一层不算毫秒。 + assert record["create_time"] == 1790574515 + # 指标按 store 的形态落成字符串,交给 ingest 的 parse_count 解析。 + assert record["liked_count"] == "3" + assert record["collected_count"] == "2" + + def test_missing_fields_do_not_crash(self): + record = douyin_api.normalize_aweme({"aweme_id": "1"}) + + assert record["aweme_id"] == "1" + assert record["title"] == "" + assert record["cover_url"] == "" + assert record["liked_count"] == "0" + assert record["create_time"] == 0 + + +class TestGet: + """`_get` 的失败路径 —— 它们决定了失败会不会被伪装成「这个博主没作品」。""" + + @staticmethod + def _client_returning(monkeypatch, status_code: int, text: str): + class _Response: + def json(self): + import json as _json + + return _json.loads(self.text) + + response = _Response() + response.status_code = status_code + response.text = text + + class _Client: + async def __aenter__(self): + return self + + async def __aexit__(self, *exc): + return False + + async def get(self, *args, **kwargs): + return response + + monkeypatch.setattr(httpx, "AsyncClient", lambda **kwargs: _Client()) + + def _identity(self): + return douyin_api.BrowserIdentity( + cookie="sessionid=s", user_agent="UA", client_hints={} + ) + + def test_an_empty_body_is_an_error_not_an_empty_result(self, monkeypatch): + """「200 + 空 body」是网关拒绝请求的典型回应。 + + 必须当场报错 —— 放过去的话,它会在下游变成「这个博主没作品」,把一次失败伪装成 + 一条正常的空结果。爬虫那条路就是这么栽的,还被翻译成「账号被封」。 + """ + self._client_returning(monkeypatch, 200, "") + + with pytest.raises(douyin_api.DouyinApiError) as excinfo: + asyncio.run(douyin_api._get("/x", {}, self._identity())) + + assert "空内容" in str(excinfo.value) + + def test_a_403_carries_the_gateways_own_message(self, monkeypatch): + """抖音难得会说原因,把它带出来,别丢。""" + self._client_returning( + monkeypatch, 403, "Blocked by ArgusSecurityPlugin Uifid Not Found" + ) + + with pytest.raises(douyin_api.DouyinApiError) as excinfo: + asyncio.run(douyin_api._get("/x", {}, self._identity())) + + assert "403" in str(excinfo.value) + assert "Uifid Not Found" in str(excinfo.value) + + def test_a_200_with_data_is_returned_as_is(self, monkeypatch): + self._client_returning(monkeypatch, 200, '{"user": {"nickname": "x"}}') + + assert asyncio.run(douyin_api._get("/x", {}, self._identity())) == { + "user": {"nickname": "x"} + }