fix(monitor): 抖音任务永远「运行中」—— page.evaluate 卡在一个死掉的标签页上,而我没给超时
用户报「一直在运行中」。库里那条 run 是真的卡住了:日志里一条采集输出都没有,
说明它卡在 collect 里、还没走到任何日志。探针定位到:
标签页: [''] ← 一个 URL 为空的标签页,渲染进程已卡死
context.cookies(): OK, 90 个 ← cookie 读得到
page.evaluate('navigator.userAgent'): **永远不返回**
而问浏览器要身份(UA + client hints)是采集的**第一步**,`page.evaluate` 又**没设超时** ——
于是整轮挂在那儿,run 永远停在「运行中」。
三处修复,各挡一层:
1. `page.evaluate` / `context.cookies()` 全部加超时(8 秒)。卡住就跳过,不再无限等。
2. 不假设第一个标签页是好的:逐个试、优先抖音页;全都不行就临时开一个干净页问完关掉。
拿不到就退回库里那份 cookie —— **不编造指纹**,那比没有更糟。
3. **进程内那条路补上整体超时**:爬虫那条靠 `run_and_wait(timeout=...)` 兜底,这条路
没有子进程、没人管,里面任何一次卡住都会变成永久的「运行中」。
测试 +6:卡死的页会被跳过(真 sleep,验的正是超时)、没 UA 的页跳过、全不行时开临时页
并关掉它、优先抖音页;以及整轮卡住时 run 不会停在 running(含超时原因)。
This commit is contained in:
@@ -116,6 +116,101 @@ class TestNormalizeAweme:
|
||||
assert record["create_time"] == 0
|
||||
|
||||
|
||||
class TestIdentityFromPages:
|
||||
"""身份得从浏览器里问,但**不能被一个卡死的标签页拖住**。
|
||||
|
||||
实测过:标签页 URL 为空、渲染进程卡死,``page.evaluate`` 永远不返回;而问身份是采集的
|
||||
第一步 —— 没超时的话整轮就挂在那儿,run 永远停在「运行中」。
|
||||
"""
|
||||
|
||||
class _Page:
|
||||
def __init__(self, url, *, user_agent=None, hang=False):
|
||||
self.url = url
|
||||
self._user_agent = user_agent
|
||||
self._hang = hang
|
||||
self.closed = False
|
||||
|
||||
async def evaluate(self, expression):
|
||||
if self._hang:
|
||||
await asyncio.sleep(30) # 模拟渲染进程卡死
|
||||
if expression.startswith("() => navigator.userAgentData"):
|
||||
return {
|
||||
"brands": [{"brand": "Chrome", "version": "155"}],
|
||||
"mobile": False,
|
||||
"platform": "Linux",
|
||||
}
|
||||
return self._user_agent
|
||||
|
||||
async def close(self):
|
||||
self.closed = True
|
||||
|
||||
class _Context:
|
||||
def __init__(self, pages, temp=None):
|
||||
self.pages = pages
|
||||
self._temp = temp
|
||||
self.made_temp = False
|
||||
|
||||
async def new_page(self):
|
||||
self.made_temp = True
|
||||
if self._temp is None:
|
||||
raise AssertionError("这个用例不该走到临时页")
|
||||
return self._temp
|
||||
|
||||
def _fast_timeout(self, monkeypatch):
|
||||
monkeypatch.setattr(douyin_api, "EVALUATE_TIMEOUT_SECONDS", 0.05)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_hanging_page_is_skipped(self, monkeypatch):
|
||||
self._fast_timeout(monkeypatch)
|
||||
stuck = self._Page("", hang=True)
|
||||
good = self._Page("https://example.com/", user_agent="UA-of-good-page")
|
||||
|
||||
user_agent, hints = await douyin_api._identity_from_pages(
|
||||
self._Context([stuck, good])
|
||||
)
|
||||
|
||||
assert user_agent == "UA-of-good-page"
|
||||
assert hints["sec-ch-ua"] == '"Chrome";v="155"'
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_page_without_a_user_agent_is_skipped(self, monkeypatch):
|
||||
self._fast_timeout(monkeypatch)
|
||||
blank = self._Page("", user_agent=None)
|
||||
good = self._Page("https://example.com/", user_agent="UA-of-good-page")
|
||||
|
||||
user_agent, _ = await douyin_api._identity_from_pages(
|
||||
self._Context([blank, good])
|
||||
)
|
||||
|
||||
assert user_agent == "UA-of-good-page"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_it_opens_a_temporary_page_when_nothing_else_works(self, monkeypatch):
|
||||
self._fast_timeout(monkeypatch)
|
||||
stuck = self._Page("", hang=True)
|
||||
temp = self._Page("about:blank", user_agent="UA-of-temp-page")
|
||||
context = self._Context([stuck], temp=temp)
|
||||
|
||||
user_agent, _ = await douyin_api._identity_from_pages(context)
|
||||
|
||||
assert context.made_temp is True
|
||||
assert user_agent == "UA-of-temp-page"
|
||||
assert temp.closed is True, "临时页问完要关掉,别在操作者的浏览器里留垃圾"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_douyin_page_is_preferred(self, monkeypatch):
|
||||
"""有抖音页面就先问它 —— 它才是我们要模仿的那个身份。"""
|
||||
self._fast_timeout(monkeypatch)
|
||||
other = self._Page("https://example.com/", user_agent="UA-of-other")
|
||||
douyin = self._Page("https://www.douyin.com/explore", user_agent="UA-of-douyin")
|
||||
|
||||
user_agent, _ = await douyin_api._identity_from_pages(
|
||||
self._Context([other, douyin])
|
||||
)
|
||||
|
||||
assert user_agent == "UA-of-douyin"
|
||||
|
||||
|
||||
class TestGet:
|
||||
"""`_get` 的失败路径 —— 它们决定了失败会不会被伪装成「这个博主没作品」。"""
|
||||
|
||||
|
||||
@@ -5,6 +5,8 @@
|
||||
约定它一个都不沾。凡是写在那里面的东西,这条路都得单独有一份。
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
|
||||
import pytest
|
||||
import pytest_asyncio
|
||||
from sqlalchemy import select
|
||||
@@ -80,3 +82,29 @@ class TestDouyinRunStatus:
|
||||
await runner_module.execute_task(task_id, trigger="manual")
|
||||
|
||||
assert seen["status"] == RUN_RUNNING
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_hanging_collect_does_not_leave_the_run_running(self, db, monkeypatch):
|
||||
"""进程内那条路也要有超时。
|
||||
|
||||
爬虫那条靠 ``run_and_wait(timeout=...)`` 兜底,这条路没有子进程、没人管 ——
|
||||
里面任何一次卡住(实测过 ``page.evaluate`` 打在一个卡死的标签页上不返回)都会让
|
||||
run 永远停在「运行中」,界面上看起来就是任务卡死了。
|
||||
"""
|
||||
task_id = await _make_douyin_task()
|
||||
async with monitor_db.get_session() as session:
|
||||
task = await session.get(MonitorTask, task_id)
|
||||
task.run_timeout_seconds = 1 # 把超时压到 1 秒,别让测试真等
|
||||
|
||||
async def hanging_collect(out_dir, **kwargs):
|
||||
await asyncio.sleep(60)
|
||||
raise AssertionError("不该走到这里")
|
||||
|
||||
monkeypatch.setattr(runner_module.douyin_fetch, "collect", hanging_collect)
|
||||
|
||||
await runner_module.execute_task(task_id, trigger="manual")
|
||||
|
||||
async with monitor_db.get_session() as session:
|
||||
run = await session.scalar(select(MonitorRun).order_by(MonitorRun.id))
|
||||
assert run.status != RUN_RUNNING
|
||||
assert "超时" in (run.error_message or "") or "超过" in (run.error_message or "")
|
||||
|
||||
Reference in New Issue
Block a user