移植自 mac-agent-os 的 mediacrawler_adapter:不起子进程、不开页面,用浏览器里那份
登录态直接调抖音 Web 接口。产物键名照抄 store/douyin,所以 ingest 那条链路一个字不用改。
**目前能用的(真环境实测,非推断)**:
profile/other : 200, 7075 字节 —— 博主主页指标(粉丝/获赞/作品数/昵称)
aweme/detail : 200, 45425 字节 —— 单条作品详情(含点赞/评论/收藏/分享)
**目前不能用的:作品列表 `aweme/post`。** 两个互相独立的原因:
1. 这个接口被抖音单独升级成了真校验:不带 x-tt-argus 回 403「Uifid Not Found」,
带上 dummy 值回 200 + **空 body**。也就是说「头在不在」骗得过,「真校验」过不了。
同一套头打 profile/other 和 aweme/detail 都是通的 —— 抖音是挑着接口加保护的,
挑中的恰好是「批量拉作品列表」这个最敏感的动作。
2. 改走页面截获也不行:CDP 浏览器打开博主主页会落到「验证码中间页」(当天大量探测的
代价,过几小时要重测)。
所以现在的边界是:**已知作品的指标刷新能做,自动发现新作品做不了**。
**排查中控住变量后得到的两条事实**(都写进注释了):
· `Accept` / `Accept-Language` / `Referer` 才是主页接口能返回真数据的原因 —— 只有
UA+client hints+Cookie 时是 200 但仅 121 字节的空壳,补上这三个头变 7074 字节。
(我先前猜的 sec-ch-ua 不是关键。)
· 因此 UA 与 client hints 必须**成套地取自同一个浏览器**,所以 BrowserIdentity 一次
从 CDP 取齐 cookie + UA + hints,而不是各自写死。
「200 + 空 body 必须当场报错」也是刻意写死的:放过去它会在下游变成「这个博主没作品」,
把一次失败伪装成一条正常结果 —— 爬虫那条路正是这么栽的,还被翻译成「账号被封」。
顺带:把参考项目目录加进 .gitignore。上一次 `git add -A` 把 mac-agent-os-main 整个
(1429 个文件)带进了提交,已从历史里清掉。
测试 +11:cookie 解析、请求头成套性(含 uifid 缺失/回退)、产物键名与 store 对齐、
以及 _get 的三条失败路径(空 body / 403 带网关原话 / 正常返回)。
182 lines
6.9 KiB
Python
182 lines
6.9 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""抖音 Web 接口客户端 —— 纯逻辑部分(不发网络请求、不连浏览器)。
|
|
|
|
发请求那半边只能在真环境里验(要 CDP 浏览器 + 登录态),所以这里钉住的是那些
|
|
「错了会一路错到入库」的地方:请求头的成套性、cookie 解析、以及产物键名。
|
|
"""
|
|
|
|
from tools.user_hash import anonymize_user_id
|
|
|
|
import asyncio
|
|
|
|
import httpx
|
|
import pytest
|
|
|
|
from api.monitor import douyin_api
|
|
|
|
|
|
class TestCookieParsing:
|
|
def test_cookie_header_is_normalised(self):
|
|
assert douyin_api._cookie_header(" a=1 ; b = 2 ;; c=3 ") == "a=1; b=2; c=3"
|
|
|
|
def test_cookie_value_lookup(self):
|
|
assert douyin_api._cookie_value("a=1; UIFID=xyz; b=2", "UIFID") == "xyz"
|
|
assert douyin_api._cookie_value("a=1", "UIFID") == ""
|
|
assert douyin_api._cookie_value("", "UIFID") == ""
|
|
|
|
def test_session_detection(self):
|
|
assert douyin_api._has_session("a=1; sessionid=abc") is True
|
|
assert douyin_api._has_session("a=1; sessionid_ss=abc") is False
|
|
assert douyin_api._has_session("") is False
|
|
|
|
|
|
class TestRequestHeaders:
|
|
"""请求头必须**成套**,而且成套地来自同一个浏览器。
|
|
|
|
实测:只有 UA + client hints + Cookie 时,主页接口回 200 但只有 121 字节(空壳);
|
|
补上 Accept / Accept-Language / Referer 才变成 7074 字节的真数据。
|
|
"""
|
|
|
|
def test_the_full_set_is_sent(self):
|
|
identity = douyin_api.BrowserIdentity(
|
|
cookie="sessionid=s; UIFID=u1",
|
|
user_agent="UA-of-this-browser",
|
|
client_hints={"sec-ch-ua": '"Chrome";v="155"'},
|
|
)
|
|
|
|
headers = identity.headers()
|
|
|
|
assert headers["User-Agent"] == "UA-of-this-browser"
|
|
assert headers["sec-ch-ua"] == '"Chrome";v="155"'
|
|
assert headers["Accept"], "Accept 系列是主页接口能不能返回真数据的必要条件"
|
|
assert headers["Accept-Language"]
|
|
assert headers["Referer"] == "https://www.douyin.com/"
|
|
assert headers["x-tt-argus"] == douyin_api.ARGUS_HEADER_VALUE
|
|
assert headers["uifid"] == "u1"
|
|
assert headers["Cookie"] == "sessionid=s; UIFID=u1"
|
|
|
|
def test_uifid_is_omitted_when_absent(self):
|
|
"""cookie 里没有 uifid 就别带 —— 送个空值反而更像异常请求。"""
|
|
identity = douyin_api.BrowserIdentity(
|
|
cookie="sessionid=s", user_agent="UA", client_hints={}
|
|
)
|
|
|
|
assert "uifid" not in identity.headers()
|
|
|
|
def test_uifid_temp_is_used_as_a_fallback(self):
|
|
identity = douyin_api.BrowserIdentity(
|
|
cookie="sessionid=s; UIFID_TEMP=temp-1", user_agent="UA", client_hints={}
|
|
)
|
|
|
|
assert identity.headers()["uifid"] == "temp-1"
|
|
|
|
|
|
class TestNormalizeAweme:
|
|
def test_keys_match_what_the_store_writes(self):
|
|
"""键名必须和 store/douyin 一模一样,否则 ingest 一条都读不到。"""
|
|
record = douyin_api.normalize_aweme(
|
|
{
|
|
"aweme_id": 7690458980574358513,
|
|
"desc": "中秋哪儿都堵",
|
|
"create_time": 1790574515,
|
|
"author": {"uid": "776719710825195", "nickname": "AA建材王总"},
|
|
"statistics": {
|
|
"digg_count": 3,
|
|
"comment_count": 1,
|
|
"collect_count": 2,
|
|
"share_count": 0,
|
|
},
|
|
"video": {"cover": {"url_list": ["https://img/cover.jpg"]}},
|
|
}
|
|
)
|
|
|
|
assert record["aweme_id"] == "7690458980574358513"
|
|
assert record["title"] == "中秋哪儿都堵"
|
|
assert record["nickname"] == "AA建材王总"
|
|
assert record["cover_url"] == "https://img/cover.jpg"
|
|
assert (
|
|
record["aweme_url"]
|
|
== "https://www.douyin.com/video/7690458980574358513"
|
|
)
|
|
# 与 store 一致:creator_hash 是 uid 的匿名哈希。
|
|
assert record["creator_hash"] == anonymize_user_id("776719710825195")
|
|
# **秒**。adapters 的 time_scale=1000 会把它换成毫秒 —— 这一层不算毫秒。
|
|
assert record["create_time"] == 1790574515
|
|
# 指标按 store 的形态落成字符串,交给 ingest 的 parse_count 解析。
|
|
assert record["liked_count"] == "3"
|
|
assert record["collected_count"] == "2"
|
|
|
|
def test_missing_fields_do_not_crash(self):
|
|
record = douyin_api.normalize_aweme({"aweme_id": "1"})
|
|
|
|
assert record["aweme_id"] == "1"
|
|
assert record["title"] == ""
|
|
assert record["cover_url"] == ""
|
|
assert record["liked_count"] == "0"
|
|
assert record["create_time"] == 0
|
|
|
|
|
|
class TestGet:
|
|
"""`_get` 的失败路径 —— 它们决定了失败会不会被伪装成「这个博主没作品」。"""
|
|
|
|
@staticmethod
|
|
def _client_returning(monkeypatch, status_code: int, text: str):
|
|
class _Response:
|
|
def json(self):
|
|
import json as _json
|
|
|
|
return _json.loads(self.text)
|
|
|
|
response = _Response()
|
|
response.status_code = status_code
|
|
response.text = text
|
|
|
|
class _Client:
|
|
async def __aenter__(self):
|
|
return self
|
|
|
|
async def __aexit__(self, *exc):
|
|
return False
|
|
|
|
async def get(self, *args, **kwargs):
|
|
return response
|
|
|
|
monkeypatch.setattr(httpx, "AsyncClient", lambda **kwargs: _Client())
|
|
|
|
def _identity(self):
|
|
return douyin_api.BrowserIdentity(
|
|
cookie="sessionid=s", user_agent="UA", client_hints={}
|
|
)
|
|
|
|
def test_an_empty_body_is_an_error_not_an_empty_result(self, monkeypatch):
|
|
"""「200 + 空 body」是网关拒绝请求的典型回应。
|
|
|
|
必须当场报错 —— 放过去的话,它会在下游变成「这个博主没作品」,把一次失败伪装成
|
|
一条正常的空结果。爬虫那条路就是这么栽的,还被翻译成「账号被封」。
|
|
"""
|
|
self._client_returning(monkeypatch, 200, "")
|
|
|
|
with pytest.raises(douyin_api.DouyinApiError) as excinfo:
|
|
asyncio.run(douyin_api._get("/x", {}, self._identity()))
|
|
|
|
assert "空内容" in str(excinfo.value)
|
|
|
|
def test_a_403_carries_the_gateways_own_message(self, monkeypatch):
|
|
"""抖音难得会说原因,把它带出来,别丢。"""
|
|
self._client_returning(
|
|
monkeypatch, 403, "Blocked by ArgusSecurityPlugin Uifid Not Found"
|
|
)
|
|
|
|
with pytest.raises(douyin_api.DouyinApiError) as excinfo:
|
|
asyncio.run(douyin_api._get("/x", {}, self._identity()))
|
|
|
|
assert "403" in str(excinfo.value)
|
|
assert "Uifid Not Found" in str(excinfo.value)
|
|
|
|
def test_a_200_with_data_is_returned_as_is(self, monkeypatch):
|
|
self._client_returning(monkeypatch, 200, '{"user": {"nickname": "x"}}')
|
|
|
|
assert asyncio.run(douyin_api._get("/x", {}, self._identity())) == {
|
|
"user": {"nickname": "x"}
|
|
}
|