fix(monitor): 抖音任务永远「运行中」—— page.evaluate 卡在一个死掉的标签页上,而我没给超时
用户报「一直在运行中」。库里那条 run 是真的卡住了:日志里一条采集输出都没有,
说明它卡在 collect 里、还没走到任何日志。探针定位到:
标签页: [''] ← 一个 URL 为空的标签页,渲染进程已卡死
context.cookies(): OK, 90 个 ← cookie 读得到
page.evaluate('navigator.userAgent'): **永远不返回**
而问浏览器要身份(UA + client hints)是采集的**第一步**,`page.evaluate` 又**没设超时** ——
于是整轮挂在那儿,run 永远停在「运行中」。
三处修复,各挡一层:
1. `page.evaluate` / `context.cookies()` 全部加超时(8 秒)。卡住就跳过,不再无限等。
2. 不假设第一个标签页是好的:逐个试、优先抖音页;全都不行就临时开一个干净页问完关掉。
拿不到就退回库里那份 cookie —— **不编造指纹**,那比没有更糟。
3. **进程内那条路补上整体超时**:爬虫那条靠 `run_and_wait(timeout=...)` 兜底,这条路
没有子进程、没人管,里面任何一次卡住都会变成永久的「运行中」。
测试 +6:卡死的页会被跳过(真 sleep,验的正是超时)、没 UA 的页跳过、全不行时开临时页
并关掉它、优先抖音页;以及整轮卡住时 run 不会停在 running(含超时原因)。
This commit is contained in:
@@ -69,6 +69,12 @@ COMMENT_PATH = "/aweme/v1/web/comment/list/"
|
||||
|
||||
# 一次请求的超时。抖音这两个接口正常都在一秒内返回。
|
||||
REQUEST_TIMEOUT_SECONDS = 20.0
|
||||
|
||||
# 问浏览器要 UA / client hints 的超时。**这个必须有。**
|
||||
# ``page.evaluate`` 打在一个渲染进程已经卡住的标签页上会**永远不返回**,而问身份是采集的
|
||||
# 第一步 —— 它一挂,整个 run 就永远停在「运行中」(真踩过:标签页 URL 是空的,
|
||||
# cookies() 正常,evaluate 一直不回来)。
|
||||
EVALUATE_TIMEOUT_SECONDS = 8.0
|
||||
# 单页最多要多少条。接口自己有上限,要多了也没用。
|
||||
MAX_PAGE_SIZE = 20
|
||||
|
||||
@@ -117,6 +123,61 @@ _IDENTITY_TTL_SECONDS = 120.0
|
||||
_identity_cache: Optional[Tuple[float, BrowserIdentity]] = None
|
||||
|
||||
|
||||
async def _safe_evaluate(page: Any, expression: str) -> Any:
|
||||
"""在页面上求值,带超时;任何失败都返回 None。
|
||||
|
||||
**不要直接调 ``page.evaluate``** —— 在渲染进程卡住的标签页上它会永远不返回(见
|
||||
``EVALUATE_TIMEOUT_SECONDS`` 那段)。
|
||||
"""
|
||||
try:
|
||||
return await asyncio.wait_for(
|
||||
page.evaluate(expression), timeout=EVALUATE_TIMEOUT_SECONDS
|
||||
)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
async def _identity_from_pages(context: Any) -> Tuple[str, Dict[str, str]]:
|
||||
"""问出 UA 和 client hints。
|
||||
|
||||
不假设第一个标签页是好的 —— 它可能停在 URL 为空、渲染进程已卡住的状态(实测过)。
|
||||
所以逐个试、每个都带超时;优先抖音页面,全都不行就临时开一个干净页问完关掉。
|
||||
|
||||
拿不到就返回空 —— 调用方据此退回库里那份 cookie,而不是拿一组编出来的指纹去请求
|
||||
(那比没有更糟,见 BrowserIdentity 的说明)。
|
||||
"""
|
||||
from media_platform.douyin.help import client_hint_headers
|
||||
|
||||
pages = list(context.pages)
|
||||
pages.sort(key=lambda page: 0 if "douyin" in (page.url or "") else 1)
|
||||
for page in pages:
|
||||
user_agent = await _safe_evaluate(page, "() => navigator.userAgent")
|
||||
if user_agent:
|
||||
hints = client_hint_headers(
|
||||
await _safe_evaluate(page, "() => navigator.userAgentData || null")
|
||||
)
|
||||
return user_agent, hints or {}
|
||||
|
||||
temp = None
|
||||
try:
|
||||
temp = await asyncio.wait_for(
|
||||
context.new_page(), timeout=EVALUATE_TIMEOUT_SECONDS
|
||||
)
|
||||
user_agent = await _safe_evaluate(temp, "() => navigator.userAgent")
|
||||
hints = client_hint_headers(
|
||||
await _safe_evaluate(temp, "() => navigator.userAgentData || null")
|
||||
)
|
||||
return user_agent or "", hints or {}
|
||||
except Exception:
|
||||
return "", {}
|
||||
finally:
|
||||
if temp is not None:
|
||||
try:
|
||||
await temp.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
async def _read_browser() -> Optional[BrowserIdentity]:
|
||||
"""连上 CDP 浏览器,一次取齐 cookie、UA、client hints。
|
||||
|
||||
@@ -134,14 +195,12 @@ async def _read_browser() -> Optional[BrowserIdentity]:
|
||||
return None
|
||||
# contexts[0] 是真实 profile。**不要 new_context()** —— 那是无痕式的,读不到登录态。
|
||||
context = browser.contexts[0]
|
||||
cookies = await context.cookies()
|
||||
cookies = await asyncio.wait_for(
|
||||
context.cookies(), timeout=EVALUATE_TIMEOUT_SECONDS
|
||||
)
|
||||
|
||||
# UA 和 hints 要从页面里问 —— 它们是浏览器自己的事实,写死迟早对不上。
|
||||
page = context.pages[0] if context.pages else await context.new_page()
|
||||
user_agent = await page.evaluate("() => navigator.userAgent")
|
||||
hints = client_hint_headers(
|
||||
await page.evaluate("() => navigator.userAgentData || null")
|
||||
)
|
||||
user_agent, hints = await _identity_from_pages(context)
|
||||
except Exception as exc:
|
||||
utils.logger.warning(f"[douyin_api] 读浏览器身份失败:{exc}")
|
||||
return None
|
||||
|
||||
+25
-11
@@ -221,17 +221,31 @@ async def execute_task(task_id: int, trigger: str = "manual") -> IngestResult:
|
||||
|
||||
in_process_tail: List[str] = []
|
||||
if platform == adapters.PLATFORM_DY:
|
||||
fetched = await douyin_fetch.collect(
|
||||
out_dir,
|
||||
platform=platform,
|
||||
mode=mode,
|
||||
limit=max_notes_count,
|
||||
want_comments=enable_comments,
|
||||
comment_limit=max_comments_count,
|
||||
targets=targets,
|
||||
known_aweme_ids=known_aweme_ids,
|
||||
cookie=cookie,
|
||||
)
|
||||
try:
|
||||
# 也要有超时。爬虫那条路靠 run_and_wait(timeout=...) 兜底,这条路没有子进程、
|
||||
# 没人管 —— 里面**任何一次卡住都会让 run 永远停在「运行中」**(真踩过:
|
||||
# page.evaluate 打在一个卡死的标签页上不返回)。
|
||||
fetched = await asyncio.wait_for(
|
||||
douyin_fetch.collect(
|
||||
out_dir,
|
||||
platform=platform,
|
||||
mode=mode,
|
||||
limit=max_notes_count,
|
||||
want_comments=enable_comments,
|
||||
comment_limit=max_comments_count,
|
||||
targets=targets,
|
||||
known_aweme_ids=known_aweme_ids,
|
||||
cookie=cookie,
|
||||
),
|
||||
timeout=timeout_seconds,
|
||||
)
|
||||
except asyncio.TimeoutError:
|
||||
fetched = {
|
||||
"notes": 0,
|
||||
"comments": 0,
|
||||
"errors": [f"抖音采集超过 {timeout_seconds} 秒仍未完成,已放弃这一轮"],
|
||||
"jsonl_dir": "",
|
||||
}
|
||||
in_process_tail = list(fetched["errors"])
|
||||
# 一条都没采到 = 这一轮失败,并把**真因**当作退出诊断传下去。否则它会掉进
|
||||
# ingest 的「疑似登录失效」分支 —— 又骗人一次,正是这套东西一直在犯的毛病。
|
||||
|
||||
Reference in New Issue
Block a user