Files
MediaCrawler/api/monitor/qrlogin.py
T
butubb 37ca1b1cd6
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
feat: CDP 接管开关 + 扫码登录面板 + Docker 部署
- runner: enable_cdp_mode 从硬编码 False 改为系统设置 cdp_enabled。服务器部署下爬虫接管已开启远程调试的 Chrome(默认 9222),复用其 profile 登录态;本机桌面默认仍为关,行为不变
- qrlogin: 新增 CDP 扫码登录。Chrome 在服务器上跑于 Xvfb,show_qrcode 依赖的 PIL 桌面看图程序不存在,二维码无处可显示;改为经 CDP 从页面取出二维码交给 WebUI 渲染。刻意复用 browser.contexts[0](新建 context 是无痕 profile,扫了也白扫),且绝不调用 browser.close()(会连带关掉操作者自己的 Chrome)
- webui: 设置页新增扫码面板,替换原本跳到采集页看终端二维码的入口
- Dockerfile / .dockerignore / docker-compose.yml: 服务器部署。host 网络是必需而非图省事——容器里 127.0.0.1:9222 必须落到宿主机回环
- UPSTREAM.md: 补充 gitcode 镜像,用于 GitHub 大包传输必断时补历史
2026-10-07 10:41:11 +08:00

269 lines
9.8 KiB
Python

# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/qrlogin.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""Show a login QR code to the operator through the WebUI.
Why this module exists: on a server deployment Chrome runs under Xvfb, so there
is no display to look at, and the helper the crawler normally uses to present the
QR (`show_qrcode` in `tools/crawler_util.py`) calls PIL's ``Image.show()`` -- it
needs a desktop image viewer that such a machine does not have, so the code would
go nowhere and the operator would be stuck. Instead the QR is read straight out
of the page over CDP and handed to the WebUI, which renders it as an ``<img>``.
Three details that are easy to get wrong:
* **Reuse the browser's default context.** ``browser.new_context()`` would give
an incognito-like profile, so the scan would land in a cookie jar the crawler
never reads and every run would still look logged out. The real profile -- the
one the crawler attaches to -- is ``browser.contexts[0]``.
* **Never call ``browser.close()``.** On a CDP connection that tears down the
operator's own Chrome and takes every unrelated tab with it. Only the page this
module opened is closed, and the Playwright client is stopped to drop the
socket.
* **A scan is not a login until the cookie changes.** The page can be showing a
QR for an account that is in fact already signed in, so success is decided by
``web_session`` appearing or changing against the value captured at start,
never by anything the page displays.
"""
import asyncio
import os
import time
from typing import Any, Dict, Optional
import config
from playwright.async_api import async_playwright
from tools import utils
from .platforms import PLATFORM_XHS
# How long a QR stays valid before the session is written off. The platform
# rotates the code well before this; the limit exists so an abandoned attempt
# cannot pin a browser tab open indefinitely.
QR_TTL_SECONDS = 180
STATUS_IDLE = "idle"
STATUS_WAITING = "waiting"
STATUS_SUCCESS = "success"
STATUS_EXPIRED = "expired"
STATUS_ERROR = "error"
# Only xhs is wired: it is the only platform whose monitor pipeline works, and
# pretending otherwise would offer the operator a button that cannot succeed.
LOGIN_URL: Dict[str, str] = {PLATFORM_XHS: "https://www.xiaohongshu.com"}
QR_SELECTOR: Dict[str, str] = {PLATFORM_XHS: "xpath=//img[@class='qrcode-img']"}
LOGIN_BUTTON_SELECTOR: Dict[str, str] = {
PLATFORM_XHS: "xpath=//*[@id='app']/div[1]/div[2]/div[1]/ul/div[1]/button"
}
SESSION_COOKIE: Dict[str, str] = {PLATFORM_XHS: "web_session"}
_IDLE_SNAPSHOT: Dict[str, Any] = {
"status": STATUS_IDLE,
"platform": None,
"image": "",
"message": "",
"elapsed": 0,
"expires_in": 0,
}
_lock = asyncio.Lock()
_current: Optional["QrLoginSession"] = None
def _cdp_url() -> str:
"""Where to reach the browser's DevTools endpoint.
``MC_CDP_URL`` wins so a deployment can point at another host without a code
change; otherwise the port comes from the same config the crawler itself
reads, so the two can never drift apart.
"""
return os.getenv("MC_CDP_URL") or f"http://127.0.0.1:{config.CDP_DEBUG_PORT}"
def _login_url(platform: str) -> str:
if platform == PLATFORM_XHS and getattr(config, "XHS_INTERNATIONAL", False):
return "https://www.rednote.com"
return LOGIN_URL[platform]
class QrLoginSession:
"""One live QR-login attempt against the CDP browser."""
def __init__(self, platform: str, playwright: Any, page: Any, baseline: str) -> None:
self.platform = platform
self.status = STATUS_WAITING
self.message = "请用小红书 App 扫描二维码"
self.image = ""
self.started_at = time.time()
self._playwright = playwright
self._page = page
# The `web_session` value present *before* the scan. An account already
# signed in has a non-empty baseline, which is why success is "changed",
# not merely "present".
self._baseline = baseline
@property
def elapsed(self) -> float:
return time.time() - self.started_at
async def refresh(self) -> None:
"""Poll the browser once for a completed scan."""
if self.status != STATUS_WAITING:
return
if self.elapsed > QR_TTL_SECONDS:
self.status = STATUS_EXPIRED
self.message = "二维码已超时,请重新获取"
return
try:
cookies = await self._page.context.cookies()
except Exception:
# The operator may have closed the tab we opened.
self.status = STATUS_ERROR
self.message = "二维码所在页面已被关闭,请重新获取"
return
token = {c["name"]: c["value"] for c in cookies}.get(
SESSION_COOKIE[self.platform], ""
)
if token and token != self._baseline:
self.status = STATUS_SUCCESS
self.message = "登录成功,登录态已写入浏览器 profile"
def snapshot(self) -> Dict[str, Any]:
return {
"status": self.status,
"platform": self.platform,
"image": self.image,
"message": self.message,
"elapsed": int(self.elapsed),
"expires_in": max(0, int(QR_TTL_SECONDS - self.elapsed)),
}
async def close(self) -> None:
"""Drop our page and the Playwright client, leaving Chrome untouched."""
try:
await self._page.close()
except Exception:
pass
try:
await self._playwright.stop()
except Exception:
pass
async def _read_qr(page: Any, platform: str) -> str:
"""Pull the QR image out of the page, opening the login dialog if needed."""
image = await utils.find_login_qrcode(page, selector=QR_SELECTOR[platform])
if image:
return image
# The dialog does not always open on its own. This is the same fallback the
# crawler's own QR flow performs before giving up.
await asyncio.sleep(0.5)
try:
await page.locator(LOGIN_BUTTON_SELECTOR[platform]).click(timeout=5000)
except Exception:
return ""
return await utils.find_login_qrcode(page, selector=QR_SELECTOR[platform])
async def _reset_locked() -> None:
global _current
if _current is not None:
await _current.close()
_current = None
async def start(platform: str = PLATFORM_XHS) -> Dict[str, Any]:
"""Open a login page in the CDP browser and return its QR code."""
if platform not in LOGIN_URL:
raise ValueError(f"平台 {platform} 尚未接入扫码登录(目前仅支持小红书)")
async with _lock:
await _reset_locked()
playwright = await async_playwright().start()
try:
browser = await playwright.chromium.connect_over_cdp(_cdp_url(), timeout=15000)
except Exception as exc:
await playwright.stop()
raise RuntimeError(
f"连接浏览器失败({_cdp_url()})。请确认服务器上的 Chrome 以 "
f"--remote-debugging-port 启动。原始错误:{exc}"
) from exc
if not browser.contexts:
await playwright.stop()
raise RuntimeError(
"浏览器没有可用上下文。CDP 已连上,但读不到 profile —— "
"请确认 Chrome 不是以无痕模式启动的。"
)
# contexts[0] is the real profile. See the module docstring.
context = browser.contexts[0]
page = await context.new_page()
try:
await page.goto(_login_url(platform), wait_until="domcontentloaded", timeout=30000)
image = await _read_qr(page, platform)
cookies = await context.cookies()
except Exception as exc:
try:
await page.close()
except Exception:
pass
await playwright.stop()
raise RuntimeError(f"打开登录页失败:{exc}") from exc
baseline = {c["name"]: c["value"] for c in cookies}.get(
SESSION_COOKIE[platform], ""
)
session = QrLoginSession(platform, playwright, page, baseline)
session.image = image
if not image:
if baseline:
session.status = STATUS_SUCCESS
session.message = "浏览器已经是登录状态,无需扫码"
else:
session.status = STATUS_ERROR
session.message = "页面上没找到二维码,请确认站点结构没有变化"
global _current
_current = session
return session.snapshot()
async def status() -> Dict[str, Any]:
async with _lock:
if _current is None:
return dict(_IDLE_SNAPSHOT)
await _current.refresh()
return _current.snapshot()
async def cancel() -> Dict[str, Any]:
async with _lock:
await _reset_locked()
snapshot = dict(_IDLE_SNAPSHOT)
snapshot["message"] = "已取消"
return snapshot
async def shutdown() -> None:
"""Release the browser tab at application shutdown."""
async with _lock:
await _reset_locked()