From 9f70cd0924b5f57fcc4c90a6677cf3e97f7c0876 Mon Sep 17 00:00:00 2001 From: butubb <1422726308@qq.com> Date: Sat, 10 Oct 2026 17:30:28 +0800 Subject: [PATCH] =?UTF-8?q?fix(monitor):=20=E6=8A=96=E9=9F=B3=E3=80=8C?= =?UTF-8?q?=E4=BD=9C=E5=93=81=E3=80=8D=E6=A8=A1=E5=BC=8F=E7=9A=84=E7=9B=AE?= =?UTF-8?q?=E6=A0=87=E8=A2=AB=E5=BD=93=E6=88=90=E5=8D=9A=E4=B8=BB=E5=8E=BB?= =?UTF-8?q?=E6=9F=A5=EF=BC=8C=E7=99=BD=E5=BA=9F=E4=B8=80=E6=9D=A1=E6=9C=AC?= =?UTF-8?q?=E6=9D=A5=E8=83=BD=E7=94=A8=E7=9A=84=E8=B7=AF?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 两种模式的**目标是不同的东西**,原来的 fetcher 却一条路走到底: * 「作品」模式(粘贴作品链接)—— 目标本身就是作品 id,直接取详情即可。**这个接口没被 那道真校验挡,今天就能用。** * 「博主」模式 —— 目标是主页 sec_uid,要先拉作品列表;那个接口被挡,退化成刷新已知作品。 原来两种都去调 author_videos(它要的是博主 sec_uid),于是「作品」模式的监控拿作品号当 sec_uid 去查,必然失败 —— 而且失败原因说得很难懂(接口回你「未登录/不是浏览器」)。 结果就是:**新建「作品」模式的抖音监控永远抓不到东西**,而那本来是现有条件下唯一能用的。 现在按 mode 分岔。测试 +2:作品模式必须走 detail 且**不得**去调列表接口(走错了会 直接抛断言);一件作品坏掉不连累其他作品。 顺带记一条排查结论:博主主页的 HTML 里**没有**作品列表(RENDER_DATA 解出来只有 {isLogin, statusCode, isSpider}),所以「走页面 HTML 免接口」那条路也是死的。 --- api/monitor/douyin_fetch.py | 61 ++++++++++++++++++++++++++----------- tests/test_douyin_fetch.py | 53 ++++++++++++++++++++++++++++++++ 2 files changed, 97 insertions(+), 17 deletions(-) diff --git a/api/monitor/douyin_fetch.py b/api/monitor/douyin_fetch.py index 0f4949c..6b49d3b 100644 --- a/api/monitor/douyin_fetch.py +++ b/api/monitor/douyin_fetch.py @@ -37,7 +37,7 @@ from typing import Any, Dict, Iterable, List, Sequence from tools import utils from . import adapters, douyin_api -from .models import MODE_CREATOR, MonitorTask +from .models import MODE_CREATOR, MODE_NOTE, MonitorTask # 拉多少条作品。接口单页上限就是 20,要多了也没用。 DEFAULT_VIDEO_LIMIT = 20 @@ -73,24 +73,22 @@ async def collect( seen_aweme: set = set() for target in targets: - sec_user_id = target.external_id + external_id = target.external_id - # 作品列表是主路径;它被那道真校验挡着时,退化到「标题/昵称靠主页接口,作品靠 - # 已知 id 逐条刷新」—— 拿不到新作品,但已知作品的指标还能继续更新。 - try: - videos = await douyin_api.author_videos( - sec_user_id, count=limit, cookie=cookie + # 两种模式的目标是不同的东西,不能走同一条路: + # 作品模式 —— 目标本身就是作品 id,直接取详情(**这个接口没被挡,今天就能用**)。 + # 博主模式 —— 目标是主页 sec_uid,要先拉作品列表;那个接口被真校验挡着,退化到 + # 刷新库里已知的作品(新作品发现不了)。 + if mode == MODE_NOTE: + try: + videos = [await douyin_api.video_detail(external_id, cookie=cookie)] + except douyin_api.DouyinApiError as exc: + errors.append(f"拉取作品 {external_id} 失败:{exc}") + videos = [] + else: + videos = await _creator_works( + external_id, limit, known_aweme_ids, seen_aweme, cookie, errors ) - except douyin_api.DouyinApiError as exc: - errors.append(f"拉取博主 {sec_user_id} 的作品列表失败:{exc}") - videos = [] - for aweme_id in known_aweme_ids: - if aweme_id in seen_aweme: - continue - try: - videos.append(await douyin_api.video_detail(aweme_id, cookie=cookie)) - except douyin_api.DouyinApiError as detail_exc: - errors.append(f"刷新作品 {aweme_id} 失败:{detail_exc}") for video in videos: aweme_id = video.get("aweme_id") @@ -118,6 +116,35 @@ async def collect( } +async def _creator_works( + sec_user_id: str, + limit: int, + known_aweme_ids: Iterable[str], + seen_aweme: set, + cookie: str, + errors: List[str], +) -> List[Dict[str, Any]]: + """一个博主的作品:先要列表,列表被挡时退化成刷新已知作品。 + + 作品列表(``aweme/post``)被抖音单独加了真校验 —— 不带 ``x-tt-argus`` 回 403, + 带上 dummy 值回 200 + 空 body。所以这里拿不到**新**作品,只能保住已知的。 + """ + try: + return await douyin_api.author_videos(sec_user_id, count=limit, cookie=cookie) + except douyin_api.DouyinApiError as exc: + errors.append(f"拉取博主 {sec_user_id} 的作品列表失败:{exc}") + + refreshed: List[Dict[str, Any]] = [] + for aweme_id in known_aweme_ids: + if aweme_id in seen_aweme: + continue + try: + refreshed.append(await douyin_api.video_detail(aweme_id, cookie=cookie)) + except douyin_api.DouyinApiError as detail_exc: + errors.append(f"刷新作品 {aweme_id} 失败:{detail_exc}") + return refreshed + + def _write_artifacts( out_dir: Path, platform: str, diff --git a/tests/test_douyin_fetch.py b/tests/test_douyin_fetch.py index fa5dbe3..172cd52 100644 --- a/tests/test_douyin_fetch.py +++ b/tests/test_douyin_fetch.py @@ -132,6 +132,59 @@ class TestHappyPath: assert result["notes"] == 1 +class TestNoteMode: + @pytest.mark.asyncio + async def test_a_work_target_is_fetched_by_detail_not_by_creator_list( + self, monkeypatch, tmp_path + ): + """作品模式的目标**本身就是作品 id**,不能拿它当博主的 sec_uid 去查列表。 + + 走错了会必然失败,而且失败原因很难看懂(接口说你没登录/不是浏览器)—— + 「粘贴作品链接的监控」今天本来是能用的,别让它因为这一处走错而废掉。 + """ + + async def must_not_be_called(*args, **kwargs): + raise AssertionError("作品模式不该去拉博主的作品列表") + + async def fake_detail(aweme_id, *, cookie=""): + return _video(aweme_id) + + monkeypatch.setattr(douyin_api, "author_videos", must_not_be_called) + monkeypatch.setattr(douyin_api, "video_detail", fake_detail) + + result = await _collect( + tmp_path, + mode="note", + want_comments=False, + targets=[_Target("111"), _Target("222")], + ) + + assert result["notes"] == 2 + assert result["errors"] == [] + + notes = _read(list((tmp_path / "douyin" / "jsonl").glob("*_contents_*.jsonl"))[0]) + assert [n["aweme_id"] for n in notes] == ["111", "222"] + + @pytest.mark.asyncio + async def test_a_broken_work_does_not_lose_the_others(self, monkeypatch, tmp_path): + async def flaky(aweme_id, *, cookie=""): + if aweme_id == "222": + raise douyin_api.DouyinApiError("作品已被删除") + return _video(aweme_id) + + monkeypatch.setattr(douyin_api, "video_detail", flaky) + + result = await _collect( + tmp_path, + mode="note", + want_comments=False, + targets=[_Target("111"), _Target("222")], + ) + + assert result["notes"] == 1 + assert any("222" in error for error in result["errors"]) + + class TestDegradation: """作品列表被挡时的行为 —— 决定了这个功能今天有没有用。"""