# -*- coding: utf-8 -*- # Copyright (c) 2025 relakkes@gmail.com # # This file is part of MediaCrawler project. # Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/qrlogin.py # GitHub: https://github.com/NanmiCoder # Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 # # 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: # 1. 不得用于任何商业用途。 # 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 # 3. 不得进行大规模爬取或对平台造成运营干扰。 # 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 # 5. 不得用于任何非法或不当的用途。 # # 详细许可条款请参阅项目根目录下的LICENSE文件。 # 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 """Show a login QR code to the operator through the WebUI. Why this module exists: on a server deployment Chrome runs under Xvfb, so there is no display to look at, and the helper the crawler normally uses to present the QR (`show_qrcode` in `tools/crawler_util.py`) calls PIL's ``Image.show()`` -- it needs a desktop image viewer that such a machine does not have, so the code would go nowhere and the operator would be stuck. Instead the QR is read straight out of the page over CDP and handed to the WebUI, which renders it as an ````. Three details that are easy to get wrong: * **Reuse the browser's default context.** ``browser.new_context()`` would give an incognito-like profile, so the scan would land in a cookie jar the crawler never reads and every run would still look logged out. The real profile -- the one the crawler attaches to -- is ``browser.contexts[0]``. * **Never call ``browser.close()``.** On a CDP connection that tears down the operator's own Chrome and takes every unrelated tab with it. Only the page this module opened is closed, and the Playwright client is stopped to drop the socket. * **A scan is not a login until the cookie changes.** The page can be showing a QR for an account that is in fact already signed in, so success is decided by ``web_session`` appearing or changing against the value captured at start, never by anything the page displays. """ import asyncio import os import time from typing import Any, Dict, Optional import config from playwright.async_api import async_playwright from tools import utils from .platforms import PLATFORM_XHS # How long a QR stays valid before the session is written off. The platform # rotates the code well before this; the limit exists so an abandoned attempt # cannot pin a browser tab open indefinitely. QR_TTL_SECONDS = 180 STATUS_IDLE = "idle" STATUS_WAITING = "waiting" STATUS_SUCCESS = "success" STATUS_EXPIRED = "expired" STATUS_ERROR = "error" # Only xhs is wired: it is the only platform whose monitor pipeline works, and # pretending otherwise would offer the operator a button that cannot succeed. LOGIN_URL: Dict[str, str] = {PLATFORM_XHS: "https://www.xiaohongshu.com"} QR_SELECTOR: Dict[str, str] = {PLATFORM_XHS: "xpath=//img[@class='qrcode-img']"} LOGIN_BUTTON_SELECTOR: Dict[str, str] = { PLATFORM_XHS: "xpath=//*[@id='app']/div[1]/div[2]/div[1]/ul/div[1]/button" } SESSION_COOKIE: Dict[str, str] = {PLATFORM_XHS: "web_session"} _IDLE_SNAPSHOT: Dict[str, Any] = { "status": STATUS_IDLE, "platform": None, "image": "", "message": "", "elapsed": 0, "expires_in": 0, } _lock = asyncio.Lock() _current: Optional["QrLoginSession"] = None def _cdp_url() -> str: """Where to reach the browser's DevTools endpoint. ``MC_CDP_URL`` wins so a deployment can point at another host without a code change; otherwise the port comes from the same config the crawler itself reads, so the two can never drift apart. """ return os.getenv("MC_CDP_URL") or f"http://127.0.0.1:{config.CDP_DEBUG_PORT}" def _login_url(platform: str) -> str: if platform == PLATFORM_XHS and getattr(config, "XHS_INTERNATIONAL", False): return "https://www.rednote.com" return LOGIN_URL[platform] class QrLoginSession: """One live QR-login attempt against the CDP browser.""" def __init__(self, platform: str, playwright: Any, page: Any, baseline: str) -> None: self.platform = platform self.status = STATUS_WAITING self.message = "请用小红书 App 扫描二维码" self.image = "" self.started_at = time.time() self._playwright = playwright self._page = page # The `web_session` value present *before* the scan. An account already # signed in has a non-empty baseline, which is why success is "changed", # not merely "present". self._baseline = baseline @property def elapsed(self) -> float: return time.time() - self.started_at async def refresh(self) -> None: """Poll the browser once for a completed scan.""" if self.status != STATUS_WAITING: return if self.elapsed > QR_TTL_SECONDS: self.status = STATUS_EXPIRED self.message = "二维码已超时,请重新获取" return try: cookies = await self._page.context.cookies() except Exception: # The operator may have closed the tab we opened. self.status = STATUS_ERROR self.message = "二维码所在页面已被关闭,请重新获取" return token = {c["name"]: c["value"] for c in cookies}.get( SESSION_COOKIE[self.platform], "" ) if token and token != self._baseline: self.status = STATUS_SUCCESS self.message = "登录成功,登录态已写入浏览器 profile" def snapshot(self) -> Dict[str, Any]: return { "status": self.status, "platform": self.platform, "image": self.image, "message": self.message, "elapsed": int(self.elapsed), "expires_in": max(0, int(QR_TTL_SECONDS - self.elapsed)), } async def close(self) -> None: """Drop our page and the Playwright client, leaving Chrome untouched.""" try: await self._page.close() except Exception: pass try: await self._playwright.stop() except Exception: pass async def _read_qr(page: Any, platform: str) -> str: """Pull the QR image out of the page, opening the login dialog if needed.""" image = await utils.find_login_qrcode(page, selector=QR_SELECTOR[platform]) if image: return image # The dialog does not always open on its own. This is the same fallback the # crawler's own QR flow performs before giving up. await asyncio.sleep(0.5) try: await page.locator(LOGIN_BUTTON_SELECTOR[platform]).click(timeout=5000) except Exception: return "" return await utils.find_login_qrcode(page, selector=QR_SELECTOR[platform]) async def _reset_locked() -> None: global _current if _current is not None: await _current.close() _current = None async def start(platform: str = PLATFORM_XHS) -> Dict[str, Any]: """Open a login page in the CDP browser and return its QR code.""" if platform not in LOGIN_URL: raise ValueError(f"平台 {platform} 尚未接入扫码登录(目前仅支持小红书)") async with _lock: await _reset_locked() playwright = await async_playwright().start() try: browser = await playwright.chromium.connect_over_cdp(_cdp_url(), timeout=15000) except Exception as exc: await playwright.stop() raise RuntimeError( f"连接浏览器失败({_cdp_url()})。请确认服务器上的 Chrome 以 " f"--remote-debugging-port 启动。原始错误:{exc}" ) from exc if not browser.contexts: await playwright.stop() raise RuntimeError( "浏览器没有可用上下文。CDP 已连上,但读不到 profile —— " "请确认 Chrome 不是以无痕模式启动的。" ) # contexts[0] is the real profile. See the module docstring. context = browser.contexts[0] page = await context.new_page() try: await page.goto(_login_url(platform), wait_until="domcontentloaded", timeout=30000) image = await _read_qr(page, platform) cookies = await context.cookies() except Exception as exc: try: await page.close() except Exception: pass await playwright.stop() raise RuntimeError(f"打开登录页失败:{exc}") from exc baseline = {c["name"]: c["value"] for c in cookies}.get( SESSION_COOKIE[platform], "" ) session = QrLoginSession(platform, playwright, page, baseline) session.image = image if not image: if baseline: session.status = STATUS_SUCCESS session.message = "浏览器已经是登录状态,无需扫码" else: session.status = STATUS_ERROR session.message = "页面上没找到二维码,请确认站点结构没有变化" global _current _current = session return session.snapshot() async def status() -> Dict[str, Any]: async with _lock: if _current is None: return dict(_IDLE_SNAPSHOT) await _current.refresh() return _current.snapshot() async def cancel() -> Dict[str, Any]: async with _lock: await _reset_locked() snapshot = dict(_IDLE_SNAPSHOT) snapshot["message"] = "已取消" return snapshot async def shutdown() -> None: """Release the browser tab at application shutdown.""" async with _lock: await _reset_locked()