# -*- coding: utf-8 -*- """抖音 Web 接口客户端 —— 纯逻辑部分(不发网络请求、不连浏览器)。 发请求那半边只能在真环境里验(要 CDP 浏览器 + 登录态),所以这里钉住的是那些 「错了会一路错到入库」的地方:请求头的成套性、cookie 解析、以及产物键名。 """ from tools.user_hash import anonymize_user_id import asyncio import httpx import pytest from api.monitor import douyin_api class TestCookieParsing: def test_cookie_header_is_normalised(self): assert douyin_api._cookie_header(" a=1 ; b = 2 ;; c=3 ") == "a=1; b=2; c=3" def test_cookie_value_lookup(self): assert douyin_api._cookie_value("a=1; UIFID=xyz; b=2", "UIFID") == "xyz" assert douyin_api._cookie_value("a=1", "UIFID") == "" assert douyin_api._cookie_value("", "UIFID") == "" def test_session_detection(self): assert douyin_api._has_session("a=1; sessionid=abc") is True assert douyin_api._has_session("a=1; sessionid_ss=abc") is False assert douyin_api._has_session("") is False class TestRequestHeaders: """请求头必须**成套**,而且成套地来自同一个浏览器。 实测:只有 UA + client hints + Cookie 时,主页接口回 200 但只有 121 字节(空壳); 补上 Accept / Accept-Language / Referer 才变成 7074 字节的真数据。 """ def test_the_full_set_is_sent(self): identity = douyin_api.BrowserIdentity( cookie="sessionid=s; UIFID=u1", user_agent="UA-of-this-browser", client_hints={"sec-ch-ua": '"Chrome";v="155"'}, ) headers = identity.headers() assert headers["User-Agent"] == "UA-of-this-browser" assert headers["sec-ch-ua"] == '"Chrome";v="155"' assert headers["Accept"], "Accept 系列是主页接口能不能返回真数据的必要条件" assert headers["Accept-Language"] assert headers["Referer"] == "https://www.douyin.com/" assert headers["x-tt-argus"] == douyin_api.ARGUS_HEADER_VALUE assert headers["uifid"] == "u1" assert headers["Cookie"] == "sessionid=s; UIFID=u1" def test_uifid_is_omitted_when_absent(self): """cookie 里没有 uifid 就别带 —— 送个空值反而更像异常请求。""" identity = douyin_api.BrowserIdentity( cookie="sessionid=s", user_agent="UA", client_hints={} ) assert "uifid" not in identity.headers() def test_uifid_temp_is_used_as_a_fallback(self): identity = douyin_api.BrowserIdentity( cookie="sessionid=s; UIFID_TEMP=temp-1", user_agent="UA", client_hints={} ) assert identity.headers()["uifid"] == "temp-1" class TestNormalizeAweme: def test_keys_match_what_the_store_writes(self): """键名必须和 store/douyin 一模一样,否则 ingest 一条都读不到。""" record = douyin_api.normalize_aweme( { "aweme_id": 7690458980574358513, "desc": "中秋哪儿都堵", "create_time": 1790574515, "author": {"uid": "776719710825195", "nickname": "AA建材王总"}, "statistics": { "digg_count": 3, "comment_count": 1, "collect_count": 2, "share_count": 0, }, "video": {"cover": {"url_list": ["https://img/cover.jpg"]}}, } ) assert record["aweme_id"] == "7690458980574358513" assert record["title"] == "中秋哪儿都堵" assert record["nickname"] == "AA建材王总" assert record["cover_url"] == "https://img/cover.jpg" assert ( record["aweme_url"] == "https://www.douyin.com/video/7690458980574358513" ) # 与 store 一致:creator_hash 是 uid 的匿名哈希。 assert record["creator_hash"] == anonymize_user_id("776719710825195") # **秒**。adapters 的 time_scale=1000 会把它换成毫秒 —— 这一层不算毫秒。 assert record["create_time"] == 1790574515 # 指标按 store 的形态落成字符串,交给 ingest 的 parse_count 解析。 assert record["liked_count"] == "3" assert record["collected_count"] == "2" def test_missing_fields_do_not_crash(self): record = douyin_api.normalize_aweme({"aweme_id": "1"}) assert record["aweme_id"] == "1" assert record["title"] == "" assert record["cover_url"] == "" assert record["liked_count"] == "0" assert record["create_time"] == 0 class TestIdentityFromPages: """身份得从浏览器里问,但**不能被一个卡死的标签页拖住**。 实测过:标签页 URL 为空、渲染进程卡死,``page.evaluate`` 永远不返回;而问身份是采集的 第一步 —— 没超时的话整轮就挂在那儿,run 永远停在「运行中」。 """ class _Page: def __init__(self, url, *, user_agent=None, hang=False): self.url = url self._user_agent = user_agent self._hang = hang self.closed = False async def evaluate(self, expression): if self._hang: await asyncio.sleep(30) # 模拟渲染进程卡死 if expression.startswith("() => navigator.userAgentData"): return { "brands": [{"brand": "Chrome", "version": "155"}], "mobile": False, "platform": "Linux", } return self._user_agent async def close(self): self.closed = True class _Context: def __init__(self, pages, temp=None): self.pages = pages self._temp = temp self.made_temp = False async def new_page(self): self.made_temp = True if self._temp is None: raise AssertionError("这个用例不该走到临时页") return self._temp def _fast_timeout(self, monkeypatch): monkeypatch.setattr(douyin_api, "EVALUATE_TIMEOUT_SECONDS", 0.05) @pytest.mark.asyncio async def test_a_hanging_page_is_skipped(self, monkeypatch): self._fast_timeout(monkeypatch) stuck = self._Page("", hang=True) good = self._Page("https://example.com/", user_agent="UA-of-good-page") user_agent, hints = await douyin_api._identity_from_pages( self._Context([stuck, good]) ) assert user_agent == "UA-of-good-page" assert hints["sec-ch-ua"] == '"Chrome";v="155"' @pytest.mark.asyncio async def test_a_page_without_a_user_agent_is_skipped(self, monkeypatch): self._fast_timeout(monkeypatch) blank = self._Page("", user_agent=None) good = self._Page("https://example.com/", user_agent="UA-of-good-page") user_agent, _ = await douyin_api._identity_from_pages( self._Context([blank, good]) ) assert user_agent == "UA-of-good-page" @pytest.mark.asyncio async def test_it_opens_a_temporary_page_when_nothing_else_works(self, monkeypatch): self._fast_timeout(monkeypatch) stuck = self._Page("", hang=True) temp = self._Page("about:blank", user_agent="UA-of-temp-page") context = self._Context([stuck], temp=temp) user_agent, _ = await douyin_api._identity_from_pages(context) assert context.made_temp is True assert user_agent == "UA-of-temp-page" assert temp.closed is True, "临时页问完要关掉,别在操作者的浏览器里留垃圾" @pytest.mark.asyncio async def test_a_douyin_page_is_preferred(self, monkeypatch): """有抖音页面就先问它 —— 它才是我们要模仿的那个身份。""" self._fast_timeout(monkeypatch) other = self._Page("https://example.com/", user_agent="UA-of-other") douyin = self._Page("https://www.douyin.com/explore", user_agent="UA-of-douyin") user_agent, _ = await douyin_api._identity_from_pages( self._Context([other, douyin]) ) assert user_agent == "UA-of-douyin" class TestGet: """`_get` 的失败路径 —— 它们决定了失败会不会被伪装成「这个博主没作品」。""" @staticmethod def _client_returning(monkeypatch, status_code: int, text: str): class _Response: def json(self): import json as _json return _json.loads(self.text) response = _Response() response.status_code = status_code response.text = text class _Client: async def __aenter__(self): return self async def __aexit__(self, *exc): return False async def get(self, *args, **kwargs): return response monkeypatch.setattr(httpx, "AsyncClient", lambda **kwargs: _Client()) def _identity(self): return douyin_api.BrowserIdentity( cookie="sessionid=s", user_agent="UA", client_hints={} ) def test_an_empty_body_is_an_error_not_an_empty_result(self, monkeypatch): """「200 + 空 body」是网关拒绝请求的典型回应。 必须当场报错 —— 放过去的话,它会在下游变成「这个博主没作品」,把一次失败伪装成 一条正常的空结果。爬虫那条路就是这么栽的,还被翻译成「账号被封」。 """ self._client_returning(monkeypatch, 200, "") with pytest.raises(douyin_api.DouyinApiError) as excinfo: asyncio.run(douyin_api._get("/x", {}, self._identity())) assert "空内容" in str(excinfo.value) def test_a_403_carries_the_gateways_own_message(self, monkeypatch): """抖音难得会说原因,把它带出来,别丢。""" self._client_returning( monkeypatch, 403, "Blocked by ArgusSecurityPlugin Uifid Not Found" ) with pytest.raises(douyin_api.DouyinApiError) as excinfo: asyncio.run(douyin_api._get("/x", {}, self._identity())) assert "403" in str(excinfo.value) assert "Uifid Not Found" in str(excinfo.value) def test_a_200_with_data_is_returned_as_is(self, monkeypatch): self._client_returning(monkeypatch, 200, '{"user": {"nickname": "x"}}') assert asyncio.run(douyin_api._get("/x", {}, self._identity())) == { "user": {"nickname": "x"} }