fix(monitor): 抖音「作品」模式的目标被当成博主去查,白废一条本来能用的路
两种模式的**目标是不同的东西**,原来的 fetcher 却一条路走到底:
* 「作品」模式(粘贴作品链接)—— 目标本身就是作品 id,直接取详情即可。**这个接口没被
那道真校验挡,今天就能用。**
* 「博主」模式 —— 目标是主页 sec_uid,要先拉作品列表;那个接口被挡,退化成刷新已知作品。
原来两种都去调 author_videos(它要的是博主 sec_uid),于是「作品」模式的监控拿作品号当
sec_uid 去查,必然失败 —— 而且失败原因说得很难懂(接口回你「未登录/不是浏览器」)。
结果就是:**新建「作品」模式的抖音监控永远抓不到东西**,而那本来是现有条件下唯一能用的。
现在按 mode 分岔。测试 +2:作品模式必须走 detail 且**不得**去调列表接口(走错了会
直接抛断言);一件作品坏掉不连累其他作品。
顺带记一条排查结论:博主主页的 HTML 里**没有**作品列表(RENDER_DATA 解出来只有
{isLogin, statusCode, isSpider}),所以「走页面 HTML 免接口」那条路也是死的。
This commit is contained in:
+44
-17
@@ -37,7 +37,7 @@ from typing import Any, Dict, Iterable, List, Sequence
|
||||
from tools import utils
|
||||
|
||||
from . import adapters, douyin_api
|
||||
from .models import MODE_CREATOR, MonitorTask
|
||||
from .models import MODE_CREATOR, MODE_NOTE, MonitorTask
|
||||
|
||||
# 拉多少条作品。接口单页上限就是 20,要多了也没用。
|
||||
DEFAULT_VIDEO_LIMIT = 20
|
||||
@@ -73,24 +73,22 @@ async def collect(
|
||||
seen_aweme: set = set()
|
||||
|
||||
for target in targets:
|
||||
sec_user_id = target.external_id
|
||||
external_id = target.external_id
|
||||
|
||||
# 作品列表是主路径;它被那道真校验挡着时,退化到「标题/昵称靠主页接口,作品靠
|
||||
# 已知 id 逐条刷新」—— 拿不到新作品,但已知作品的指标还能继续更新。
|
||||
try:
|
||||
videos = await douyin_api.author_videos(
|
||||
sec_user_id, count=limit, cookie=cookie
|
||||
# 两种模式的目标是不同的东西,不能走同一条路:
|
||||
# 作品模式 —— 目标本身就是作品 id,直接取详情(**这个接口没被挡,今天就能用**)。
|
||||
# 博主模式 —— 目标是主页 sec_uid,要先拉作品列表;那个接口被真校验挡着,退化到
|
||||
# 刷新库里已知的作品(新作品发现不了)。
|
||||
if mode == MODE_NOTE:
|
||||
try:
|
||||
videos = [await douyin_api.video_detail(external_id, cookie=cookie)]
|
||||
except douyin_api.DouyinApiError as exc:
|
||||
errors.append(f"拉取作品 {external_id} 失败:{exc}")
|
||||
videos = []
|
||||
else:
|
||||
videos = await _creator_works(
|
||||
external_id, limit, known_aweme_ids, seen_aweme, cookie, errors
|
||||
)
|
||||
except douyin_api.DouyinApiError as exc:
|
||||
errors.append(f"拉取博主 {sec_user_id} 的作品列表失败:{exc}")
|
||||
videos = []
|
||||
for aweme_id in known_aweme_ids:
|
||||
if aweme_id in seen_aweme:
|
||||
continue
|
||||
try:
|
||||
videos.append(await douyin_api.video_detail(aweme_id, cookie=cookie))
|
||||
except douyin_api.DouyinApiError as detail_exc:
|
||||
errors.append(f"刷新作品 {aweme_id} 失败:{detail_exc}")
|
||||
|
||||
for video in videos:
|
||||
aweme_id = video.get("aweme_id")
|
||||
@@ -118,6 +116,35 @@ async def collect(
|
||||
}
|
||||
|
||||
|
||||
async def _creator_works(
|
||||
sec_user_id: str,
|
||||
limit: int,
|
||||
known_aweme_ids: Iterable[str],
|
||||
seen_aweme: set,
|
||||
cookie: str,
|
||||
errors: List[str],
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""一个博主的作品:先要列表,列表被挡时退化成刷新已知作品。
|
||||
|
||||
作品列表(``aweme/post``)被抖音单独加了真校验 —— 不带 ``x-tt-argus`` 回 403,
|
||||
带上 dummy 值回 200 + 空 body。所以这里拿不到**新**作品,只能保住已知的。
|
||||
"""
|
||||
try:
|
||||
return await douyin_api.author_videos(sec_user_id, count=limit, cookie=cookie)
|
||||
except douyin_api.DouyinApiError as exc:
|
||||
errors.append(f"拉取博主 {sec_user_id} 的作品列表失败:{exc}")
|
||||
|
||||
refreshed: List[Dict[str, Any]] = []
|
||||
for aweme_id in known_aweme_ids:
|
||||
if aweme_id in seen_aweme:
|
||||
continue
|
||||
try:
|
||||
refreshed.append(await douyin_api.video_detail(aweme_id, cookie=cookie))
|
||||
except douyin_api.DouyinApiError as detail_exc:
|
||||
errors.append(f"刷新作品 {aweme_id} 失败:{detail_exc}")
|
||||
return refreshed
|
||||
|
||||
|
||||
def _write_artifacts(
|
||||
out_dir: Path,
|
||||
platform: str,
|
||||
|
||||
@@ -132,6 +132,59 @@ class TestHappyPath:
|
||||
assert result["notes"] == 1
|
||||
|
||||
|
||||
class TestNoteMode:
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_work_target_is_fetched_by_detail_not_by_creator_list(
|
||||
self, monkeypatch, tmp_path
|
||||
):
|
||||
"""作品模式的目标**本身就是作品 id**,不能拿它当博主的 sec_uid 去查列表。
|
||||
|
||||
走错了会必然失败,而且失败原因很难看懂(接口说你没登录/不是浏览器)——
|
||||
「粘贴作品链接的监控」今天本来是能用的,别让它因为这一处走错而废掉。
|
||||
"""
|
||||
|
||||
async def must_not_be_called(*args, **kwargs):
|
||||
raise AssertionError("作品模式不该去拉博主的作品列表")
|
||||
|
||||
async def fake_detail(aweme_id, *, cookie=""):
|
||||
return _video(aweme_id)
|
||||
|
||||
monkeypatch.setattr(douyin_api, "author_videos", must_not_be_called)
|
||||
monkeypatch.setattr(douyin_api, "video_detail", fake_detail)
|
||||
|
||||
result = await _collect(
|
||||
tmp_path,
|
||||
mode="note",
|
||||
want_comments=False,
|
||||
targets=[_Target("111"), _Target("222")],
|
||||
)
|
||||
|
||||
assert result["notes"] == 2
|
||||
assert result["errors"] == []
|
||||
|
||||
notes = _read(list((tmp_path / "douyin" / "jsonl").glob("*_contents_*.jsonl"))[0])
|
||||
assert [n["aweme_id"] for n in notes] == ["111", "222"]
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_broken_work_does_not_lose_the_others(self, monkeypatch, tmp_path):
|
||||
async def flaky(aweme_id, *, cookie=""):
|
||||
if aweme_id == "222":
|
||||
raise douyin_api.DouyinApiError("作品已被删除")
|
||||
return _video(aweme_id)
|
||||
|
||||
monkeypatch.setattr(douyin_api, "video_detail", flaky)
|
||||
|
||||
result = await _collect(
|
||||
tmp_path,
|
||||
mode="note",
|
||||
want_comments=False,
|
||||
targets=[_Target("111"), _Target("222")],
|
||||
)
|
||||
|
||||
assert result["notes"] == 1
|
||||
assert any("222" in error for error in result["errors"])
|
||||
|
||||
|
||||
class TestDegradation:
|
||||
"""作品列表被挡时的行为 —— 决定了这个功能今天有没有用。"""
|
||||
|
||||
|
||||
Reference in New Issue
Block a user