用户报「一直在运行中」。库里那条 run 是真的卡住了:日志里一条采集输出都没有,
说明它卡在 collect 里、还没走到任何日志。探针定位到:
标签页: [''] ← 一个 URL 为空的标签页,渲染进程已卡死
context.cookies(): OK, 90 个 ← cookie 读得到
page.evaluate('navigator.userAgent'): **永远不返回**
而问浏览器要身份(UA + client hints)是采集的**第一步**,`page.evaluate` 又**没设超时** ——
于是整轮挂在那儿,run 永远停在「运行中」。
三处修复,各挡一层:
1. `page.evaluate` / `context.cookies()` 全部加超时(8 秒)。卡住就跳过,不再无限等。
2. 不假设第一个标签页是好的:逐个试、优先抖音页;全都不行就临时开一个干净页问完关掉。
拿不到就退回库里那份 cookie —— **不编造指纹**,那比没有更糟。
3. **进程内那条路补上整体超时**:爬虫那条靠 `run_and_wait(timeout=...)` 兜底,这条路
没有子进程、没人管,里面任何一次卡住都会变成永久的「运行中」。
测试 +6:卡死的页会被跳过(真 sleep,验的正是超时)、没 UA 的页跳过、全不行时开临时页
并关掉它、优先抖音页;以及整轮卡住时 run 不会停在 running(含超时原因)。
277 lines
10 KiB
Python
277 lines
10 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""抖音 Web 接口客户端 —— 纯逻辑部分(不发网络请求、不连浏览器)。
|
|
|
|
发请求那半边只能在真环境里验(要 CDP 浏览器 + 登录态),所以这里钉住的是那些
|
|
「错了会一路错到入库」的地方:请求头的成套性、cookie 解析、以及产物键名。
|
|
"""
|
|
|
|
from tools.user_hash import anonymize_user_id
|
|
|
|
import asyncio
|
|
|
|
import httpx
|
|
import pytest
|
|
|
|
from api.monitor import douyin_api
|
|
|
|
|
|
class TestCookieParsing:
|
|
def test_cookie_header_is_normalised(self):
|
|
assert douyin_api._cookie_header(" a=1 ; b = 2 ;; c=3 ") == "a=1; b=2; c=3"
|
|
|
|
def test_cookie_value_lookup(self):
|
|
assert douyin_api._cookie_value("a=1; UIFID=xyz; b=2", "UIFID") == "xyz"
|
|
assert douyin_api._cookie_value("a=1", "UIFID") == ""
|
|
assert douyin_api._cookie_value("", "UIFID") == ""
|
|
|
|
def test_session_detection(self):
|
|
assert douyin_api._has_session("a=1; sessionid=abc") is True
|
|
assert douyin_api._has_session("a=1; sessionid_ss=abc") is False
|
|
assert douyin_api._has_session("") is False
|
|
|
|
|
|
class TestRequestHeaders:
|
|
"""请求头必须**成套**,而且成套地来自同一个浏览器。
|
|
|
|
实测:只有 UA + client hints + Cookie 时,主页接口回 200 但只有 121 字节(空壳);
|
|
补上 Accept / Accept-Language / Referer 才变成 7074 字节的真数据。
|
|
"""
|
|
|
|
def test_the_full_set_is_sent(self):
|
|
identity = douyin_api.BrowserIdentity(
|
|
cookie="sessionid=s; UIFID=u1",
|
|
user_agent="UA-of-this-browser",
|
|
client_hints={"sec-ch-ua": '"Chrome";v="155"'},
|
|
)
|
|
|
|
headers = identity.headers()
|
|
|
|
assert headers["User-Agent"] == "UA-of-this-browser"
|
|
assert headers["sec-ch-ua"] == '"Chrome";v="155"'
|
|
assert headers["Accept"], "Accept 系列是主页接口能不能返回真数据的必要条件"
|
|
assert headers["Accept-Language"]
|
|
assert headers["Referer"] == "https://www.douyin.com/"
|
|
assert headers["x-tt-argus"] == douyin_api.ARGUS_HEADER_VALUE
|
|
assert headers["uifid"] == "u1"
|
|
assert headers["Cookie"] == "sessionid=s; UIFID=u1"
|
|
|
|
def test_uifid_is_omitted_when_absent(self):
|
|
"""cookie 里没有 uifid 就别带 —— 送个空值反而更像异常请求。"""
|
|
identity = douyin_api.BrowserIdentity(
|
|
cookie="sessionid=s", user_agent="UA", client_hints={}
|
|
)
|
|
|
|
assert "uifid" not in identity.headers()
|
|
|
|
def test_uifid_temp_is_used_as_a_fallback(self):
|
|
identity = douyin_api.BrowserIdentity(
|
|
cookie="sessionid=s; UIFID_TEMP=temp-1", user_agent="UA", client_hints={}
|
|
)
|
|
|
|
assert identity.headers()["uifid"] == "temp-1"
|
|
|
|
|
|
class TestNormalizeAweme:
|
|
def test_keys_match_what_the_store_writes(self):
|
|
"""键名必须和 store/douyin 一模一样,否则 ingest 一条都读不到。"""
|
|
record = douyin_api.normalize_aweme(
|
|
{
|
|
"aweme_id": 7690458980574358513,
|
|
"desc": "中秋哪儿都堵",
|
|
"create_time": 1790574515,
|
|
"author": {"uid": "776719710825195", "nickname": "AA建材王总"},
|
|
"statistics": {
|
|
"digg_count": 3,
|
|
"comment_count": 1,
|
|
"collect_count": 2,
|
|
"share_count": 0,
|
|
},
|
|
"video": {"cover": {"url_list": ["https://img/cover.jpg"]}},
|
|
}
|
|
)
|
|
|
|
assert record["aweme_id"] == "7690458980574358513"
|
|
assert record["title"] == "中秋哪儿都堵"
|
|
assert record["nickname"] == "AA建材王总"
|
|
assert record["cover_url"] == "https://img/cover.jpg"
|
|
assert (
|
|
record["aweme_url"]
|
|
== "https://www.douyin.com/video/7690458980574358513"
|
|
)
|
|
# 与 store 一致:creator_hash 是 uid 的匿名哈希。
|
|
assert record["creator_hash"] == anonymize_user_id("776719710825195")
|
|
# **秒**。adapters 的 time_scale=1000 会把它换成毫秒 —— 这一层不算毫秒。
|
|
assert record["create_time"] == 1790574515
|
|
# 指标按 store 的形态落成字符串,交给 ingest 的 parse_count 解析。
|
|
assert record["liked_count"] == "3"
|
|
assert record["collected_count"] == "2"
|
|
|
|
def test_missing_fields_do_not_crash(self):
|
|
record = douyin_api.normalize_aweme({"aweme_id": "1"})
|
|
|
|
assert record["aweme_id"] == "1"
|
|
assert record["title"] == ""
|
|
assert record["cover_url"] == ""
|
|
assert record["liked_count"] == "0"
|
|
assert record["create_time"] == 0
|
|
|
|
|
|
class TestIdentityFromPages:
|
|
"""身份得从浏览器里问,但**不能被一个卡死的标签页拖住**。
|
|
|
|
实测过:标签页 URL 为空、渲染进程卡死,``page.evaluate`` 永远不返回;而问身份是采集的
|
|
第一步 —— 没超时的话整轮就挂在那儿,run 永远停在「运行中」。
|
|
"""
|
|
|
|
class _Page:
|
|
def __init__(self, url, *, user_agent=None, hang=False):
|
|
self.url = url
|
|
self._user_agent = user_agent
|
|
self._hang = hang
|
|
self.closed = False
|
|
|
|
async def evaluate(self, expression):
|
|
if self._hang:
|
|
await asyncio.sleep(30) # 模拟渲染进程卡死
|
|
if expression.startswith("() => navigator.userAgentData"):
|
|
return {
|
|
"brands": [{"brand": "Chrome", "version": "155"}],
|
|
"mobile": False,
|
|
"platform": "Linux",
|
|
}
|
|
return self._user_agent
|
|
|
|
async def close(self):
|
|
self.closed = True
|
|
|
|
class _Context:
|
|
def __init__(self, pages, temp=None):
|
|
self.pages = pages
|
|
self._temp = temp
|
|
self.made_temp = False
|
|
|
|
async def new_page(self):
|
|
self.made_temp = True
|
|
if self._temp is None:
|
|
raise AssertionError("这个用例不该走到临时页")
|
|
return self._temp
|
|
|
|
def _fast_timeout(self, monkeypatch):
|
|
monkeypatch.setattr(douyin_api, "EVALUATE_TIMEOUT_SECONDS", 0.05)
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_a_hanging_page_is_skipped(self, monkeypatch):
|
|
self._fast_timeout(monkeypatch)
|
|
stuck = self._Page("", hang=True)
|
|
good = self._Page("https://example.com/", user_agent="UA-of-good-page")
|
|
|
|
user_agent, hints = await douyin_api._identity_from_pages(
|
|
self._Context([stuck, good])
|
|
)
|
|
|
|
assert user_agent == "UA-of-good-page"
|
|
assert hints["sec-ch-ua"] == '"Chrome";v="155"'
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_a_page_without_a_user_agent_is_skipped(self, monkeypatch):
|
|
self._fast_timeout(monkeypatch)
|
|
blank = self._Page("", user_agent=None)
|
|
good = self._Page("https://example.com/", user_agent="UA-of-good-page")
|
|
|
|
user_agent, _ = await douyin_api._identity_from_pages(
|
|
self._Context([blank, good])
|
|
)
|
|
|
|
assert user_agent == "UA-of-good-page"
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_it_opens_a_temporary_page_when_nothing_else_works(self, monkeypatch):
|
|
self._fast_timeout(monkeypatch)
|
|
stuck = self._Page("", hang=True)
|
|
temp = self._Page("about:blank", user_agent="UA-of-temp-page")
|
|
context = self._Context([stuck], temp=temp)
|
|
|
|
user_agent, _ = await douyin_api._identity_from_pages(context)
|
|
|
|
assert context.made_temp is True
|
|
assert user_agent == "UA-of-temp-page"
|
|
assert temp.closed is True, "临时页问完要关掉,别在操作者的浏览器里留垃圾"
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_a_douyin_page_is_preferred(self, monkeypatch):
|
|
"""有抖音页面就先问它 —— 它才是我们要模仿的那个身份。"""
|
|
self._fast_timeout(monkeypatch)
|
|
other = self._Page("https://example.com/", user_agent="UA-of-other")
|
|
douyin = self._Page("https://www.douyin.com/explore", user_agent="UA-of-douyin")
|
|
|
|
user_agent, _ = await douyin_api._identity_from_pages(
|
|
self._Context([other, douyin])
|
|
)
|
|
|
|
assert user_agent == "UA-of-douyin"
|
|
|
|
|
|
class TestGet:
|
|
"""`_get` 的失败路径 —— 它们决定了失败会不会被伪装成「这个博主没作品」。"""
|
|
|
|
@staticmethod
|
|
def _client_returning(monkeypatch, status_code: int, text: str):
|
|
class _Response:
|
|
def json(self):
|
|
import json as _json
|
|
|
|
return _json.loads(self.text)
|
|
|
|
response = _Response()
|
|
response.status_code = status_code
|
|
response.text = text
|
|
|
|
class _Client:
|
|
async def __aenter__(self):
|
|
return self
|
|
|
|
async def __aexit__(self, *exc):
|
|
return False
|
|
|
|
async def get(self, *args, **kwargs):
|
|
return response
|
|
|
|
monkeypatch.setattr(httpx, "AsyncClient", lambda **kwargs: _Client())
|
|
|
|
def _identity(self):
|
|
return douyin_api.BrowserIdentity(
|
|
cookie="sessionid=s", user_agent="UA", client_hints={}
|
|
)
|
|
|
|
def test_an_empty_body_is_an_error_not_an_empty_result(self, monkeypatch):
|
|
"""「200 + 空 body」是网关拒绝请求的典型回应。
|
|
|
|
必须当场报错 —— 放过去的话,它会在下游变成「这个博主没作品」,把一次失败伪装成
|
|
一条正常的空结果。爬虫那条路就是这么栽的,还被翻译成「账号被封」。
|
|
"""
|
|
self._client_returning(monkeypatch, 200, "")
|
|
|
|
with pytest.raises(douyin_api.DouyinApiError) as excinfo:
|
|
asyncio.run(douyin_api._get("/x", {}, self._identity()))
|
|
|
|
assert "空内容" in str(excinfo.value)
|
|
|
|
def test_a_403_carries_the_gateways_own_message(self, monkeypatch):
|
|
"""抖音难得会说原因,把它带出来,别丢。"""
|
|
self._client_returning(
|
|
monkeypatch, 403, "Blocked by ArgusSecurityPlugin Uifid Not Found"
|
|
)
|
|
|
|
with pytest.raises(douyin_api.DouyinApiError) as excinfo:
|
|
asyncio.run(douyin_api._get("/x", {}, self._identity()))
|
|
|
|
assert "403" in str(excinfo.value)
|
|
assert "Uifid Not Found" in str(excinfo.value)
|
|
|
|
def test_a_200_with_data_is_returned_as_is(self, monkeypatch):
|
|
self._client_returning(monkeypatch, 200, '{"user": {"nickname": "x"}}')
|
|
|
|
assert asyncio.run(douyin_api._get("/x", {}, self._identity())) == {
|
|
"user": {"nickname": "x"}
|
|
}
|