fix(douyin): 缺 sec-ch-ua 请求头,网关回 200 + 空 body(不是「账号被封」)

先纠正一个我上一轮给错的结论:日志里的 `account blocked` **不是抖音说的**,是
MediaCrawler 自己编的:

    if response.text == "" or response.text == "blocked":
        raise Exception("account blocked")

真实情况只是**抖音返回了空 body**。我把它读成了「账号被风控」,还写进了运行历史和
给用户的结论里 —— 用户质疑「我网页版和手机版都能正常登录」,一查,他是对的。

实测定位(同一 URL、同一 cookie、同一参数):

  浏览器页面内 fetch : 200, 7077 字节  ✓
  httpx              : 200,    0 字节  ✗
    带 a_bogus       : 0 字节
    不带 a_bogus     : 0 字节
    四种 msToken 变体 : 全部 200 有数据(所以不是它)
  用浏览器那次的完整头重放 httpx : 200, 7077 字节 ✓

浏览器那次请求比爬虫多的,只有这三个头:

    sec-ch-ua: "Google Chrome";v="155", "Chromium";v="155", "Not(A:Brand";v=24
    sec-ch-ua-mobile: ?0
    sec-ch-ua-platform: "Linux"

爬虫的 UA 是从页面读的(声称是 Chrome 155)却不带 sec-ch-ua —— 「Chrome 的 UA +
没有 sec-ch-ua」是最典型的机器人特征。网关的回应方式是不报错、不给原因,回一个
200 + 空 body,HTTP 状态还写在成功那一栏。

修:media_platform/douyin/help.py 新增 client_hint_headers(),从
navigator.userAgentData 现算这三个头(现算而不是写死 —— 写死的版本号一旦和 UA 里的
对不上,就又是一个可疑特征);core.py 建客户端时带上。

诚实说明:我无法解释**为什么之前能跑**(run 34 还是成功的,40 分钟后同样的代码就
不行了)。最可能是字节那边收紧了这道校验,但我没有证据,别当结论。

测试 +8:还原出的头与真实浏览器抓到的值逐字一致;拿不到 userAgentData 时返回空而不
凭空编造(编一组和 UA 对不上的头比不带头更糟);mobile 标志;platform 缺失时仍发另两个。
This commit is contained in:
2026-10-10 16:47:49 +08:00
parent e4affe9170
commit 21ff01b894
3 changed files with 100 additions and 2 deletions
+12 -1
View File
@@ -44,7 +44,11 @@ from . import media as douyin_media
from .client import DouYinClient
from .exception import DataFetchError
from .field import PublishTimeType
from .help import parse_video_info_from_url, parse_creator_info_from_url
from .help import (
client_hint_headers,
parse_creator_info_from_url,
parse_video_info_from_url,
)
from .login import DouYinLogin
@@ -320,10 +324,17 @@ class DouYinCrawler(AbstractCrawler):
self.browser_context,
urls=self.cookie_urls,
) # type: ignore
# 声称自己是 Chrome,就得带上 sec-ch-ua 系列头 —— 浏览器一定会带,而缺了它们
# 的请求在抖音网关看来就是机器人:回一个 **200 + 空 body**,不报错、不给原因,
# 表现为采集抓到 0 条。见 help.client_hint_headers 的实测记录。
client_hints = client_hint_headers(
await self.context_page.evaluate("() => navigator.userAgentData || null")
)
douyin_client = DouYinClient(
proxy=httpx_proxy,
headers={
"User-Agent": await self.context_page.evaluate("() => navigator.userAgent"),
**client_hints,
"Cookie": cookie_str,
"Host": "www.douyin.com",
"Origin": "https://www.douyin.com/",
+33 -1
View File
@@ -26,7 +26,7 @@
import random
import re
from typing import Optional
from typing import Dict, Optional
import execjs
from playwright.async_api import Page
@@ -98,6 +98,38 @@ async def get_a_bogus_from_playwright(params: str, post_data: dict, user_agent:
return a_bogus
def client_hint_headers(user_agent_data) -> Dict[str, str]:
"""由 ``navigator.userAgentData`` 还原 ``sec-ch-ua`` 系列请求头。
浏览器只要声称自己是 Chrome,就**一定会**带这三个头。缺了它们,「Chrome 的 UA +
没有 sec-ch-ua」就是最典型的机器人特征 —— 抖音网关会因此返回 **200 + 空 body**:
不报错、不给原因、HTTP 状态还是成功的,表现为采集拿到 0 条。
实测(同一 URL、同一 cookie、同一参数):不带头 → 0 字节;补上这三个头 → 7077 字节。
从 ``userAgentData`` 现算而不是写死,是为了 Chrome 升级后不会悄悄失配 —— 写死的
版本号和 UA 里的版本号一旦对不上,就又是一个可疑特征。
"""
if not isinstance(user_agent_data, dict):
return {}
brands = user_agent_data.get("brands") or []
sec_ch_ua = ", ".join(
f'"{brand.get("brand", "")}";v="{brand.get("version", "")}"' for brand in brands
)
if not sec_ch_ua:
return {}
headers = {
"sec-ch-ua": sec_ch_ua,
"sec-ch-ua-mobile": "?1" if user_agent_data.get("mobile") else "?0",
}
platform = user_agent_data.get("platform")
if platform:
headers["sec-ch-ua-platform"] = f'"{platform}"'
return headers
def parse_video_info_from_url(url: str) -> VideoUrlInfo:
"""
Parse video ID from Douyin video URL
+55
View File
@@ -0,0 +1,55 @@
# -*- coding: utf-8 -*-
"""sec-ch-ua 系列请求头:为什么必须带、怎么还原。
抖音网关对「声称自己是 Chrome、却没带 sec-ch-ua」的请求会回 **200 + 空 body** ——
不报错、不给原因、HTTP 状态还是成功的,采集侧只看到 0 条。实测同一 URL、同一 cookie、
同一参数:不带头 0 字节,补上这三个头 7077 字节。
"""
import pytest
from media_platform.douyin.help import client_hint_headers
# 实测从真实浏览器抓到的 navigator.userAgentData
REAL_UA_DATA = {
"brands": [
{"brand": "Google Chrome", "version": "155"},
{"brand": "Chromium", "version": "155"},
{"brand": "Not(A:Brand", "version": "24"},
],
"mobile": False,
"platform": "Linux",
}
def test_headers_are_derived_from_user_agent_data():
hints = client_hint_headers(REAL_UA_DATA)
assert hints["sec-ch-ua"] == (
'"Google Chrome";v="155", "Chromium";v="155", "Not(A:Brand";v="24"'
)
assert hints["sec-ch-ua-mobile"] == "?0"
assert hints["sec-ch-ua-platform"] == '"Linux"'
def test_mobile_is_reflected():
hints = client_hint_headers({**REAL_UA_DATA, "mobile": True})
assert hints["sec-ch-ua-mobile"] == "?1"
@pytest.mark.parametrize("value", [None, [], "not-a-dict", {}, {"brands": []}])
def test_nothing_is_invented_when_user_agent_data_is_unavailable(value):
"""拿不到就返回空。
凭空造一组和 UA 对不上的头,只会变成**另一个**可疑特征 —— 那比不带头更糟。
"""
assert client_hint_headers(value) == {}
def test_missing_platform_still_sends_the_other_two():
hints = client_hint_headers({"brands": REAL_UA_DATA["brands"], "mobile": False})
assert "sec-ch-ua" in hints
assert "sec-ch-ua-mobile" in hints
assert "sec-ch-ua-platform" not in hints