Files
MediaCrawler/api/creator/login.py
T
butubb bef0a4fbde
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
fix(creator): 扫码成功却一直停在二维码上 + 运营改为左右布局
【扫码不完成】根因是判据本身。原先读页面里的 window.__INITIAL_STATE__,而那是
页面加载那一刻的快照:监控那边的同一探针能用,是因为那台浏览器的页面加载时就已经
登录了;而扫码是加载之后才登录的 —— SPA 内部确实登进去了,但初始快照不会翻转,
于是检测永远等不到。

改成拿 cookie 直接问创作者后台 /api/galaxy/user/info 我是谁。实测这个判据很干净:
游客也会拿到 a1(所以签名算得出来),但接口直接回 401 无登录信息;只有真正登录了
才返回 user_id。所以「有 a1」什么都证明不了,后台认了才算。
顺带按 5 秒节流 —— 前端每 2 秒问一次,没必要每次都打后台接口。

【弹窗不关】成功后不自动关闭,停在二维码上会让人以为没成功。现在显示账号卡片与原话
提示,1.6 秒后自动关闭并提供一个「完成」按钮。

【已完成结果会残留】take_cookie 取走 cookie 就拆会话,而在飞的轮询会看到 _current 为空
回报 idle,把已显示的成功能擦掉。现在把结果记在模块里重复返回,关闭弹窗时清掉 ——
否则下次打开会立刻显示上次的成功。

【布局】按用户要求改成左右两栏(左账号列表、右数据面板),与监控统一,取消二级菜单。
未选过时默认选中第一个,右栏不会一开始就是空的。

测试:tests/test_creator_login.py 新增 7 例,含「游客会话永不完成」「接口不打满每次轮询」
「临时上下文用完必须关掉」。
2026-10-07 16:40:24 +08:00

301 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/creator/login.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""运营账号的扫码登录。
**与监控的扫码登录(`api.monitor.qrlogin`)有一处决定性差异**:那边把登录态写进
浏览器**默认 profile**,因为爬虫要复用它;这边要的是 **cookie 字符串**,因为采集
走纯 HTTP。所以这里每次登录都开一个**临时上下文**,扫完取出 cookie 就丢弃 ——
* 登第二个账号不会把第一个顶掉(默认 profile 只能装一个登录态);
* 完全不影响监控那个登录态;
* 十个账号互不干扰。
扫码入口仍是主站(`www.xiaohongshu.com`):Phase 0 实测证明**主站的 cookie 就能
认证创作者后台**,不需要单独的创作者登录。
"""
import asyncio
import time
from typing import Any, Dict, Optional
from playwright.async_api import async_playwright
from tools import utils
from ..monitor.platforms import PLATFORM_XHS
from .client import CreatorApiError, CreatorClient
QR_TTL_SECONDS = 300
STATUS_IDLE = "idle"
STATUS_WAITING = "waiting"
STATUS_SUCCESS = "success"
STATUS_EXPIRED = "expired"
STATUS_ERROR = "error"
LOGIN_URL = "https://www.xiaohongshu.com"
QR_SELECTOR = "xpath=//img[@class='qrcode-img']"
LOGIN_BUTTON_SELECTOR = "xpath=//*[@id='app']/div[1]/div[2]/div[1]/ul/div[1]/button"
# 判据不读页面状态,而是拿 cookie 直接问创作者后台"我是谁"。
#
# **为什么不用页面状态**:`window.__INITIAL_STATE__` 是**页面加载那一刻的快照**。
# 监控那边的同一个探针能用,是因为那台浏览器的页面加载时就已经登录了,快照里
# loggedIn 就是 true。而扫码是"页面加载之后才登录的"—— SPA 内部确实登进去了,
# 但那个初始快照不会翻转,于是检测永远等不到,界面就一直停在二维码上。
#
# `user/info` 则是权威的:实测**游客也会拿到 a1**(所以签名算得出来),但接口直接
# 回 401「无登录信息」;只有真正登录了才返回 user_id。所以"有 a1"什么都证明不了,
# "后台认这份身份"才是。
LOGIN_CHECK_INTERVAL_SECONDS = 5.0
_lock = asyncio.Lock()
_current: Optional["AccountLoginSession"] = None
# 完成后的快照。会话一旦被取走 cookie 就会拆掉,而前端可能还有一个在飞的轮询——
# 那个请求若看到 _current 为空就会回报 idle,把已经显示出来的成功状态又擦掉。
# 把结果留在这里,重复轮询就稳定得多。
_last_result: Optional[Dict[str, Any]] = None
_playwright: Any = None
def _cdp_url() -> str:
import os
import config
return os.getenv("MC_CDP_URL") or f"http://127.0.0.1:{config.CDP_DEBUG_PORT}"
async def _connect():
global _playwright
if _playwright is None:
_playwright = await async_playwright().start()
return await _playwright.chromium.connect_over_cdp(_cdp_url(), timeout=15000)
async def _disconnect() -> None:
global _playwright
if _playwright is not None:
try:
await _playwright.stop()
except Exception:
pass
_playwright = None
async def _read_qr(page: Any) -> str:
image = await utils.find_login_qrcode(page, selector=QR_SELECTOR)
if image:
return image
# 登录框不一定自己弹出来,这是爬虫自身扫码流程的同款兜底。
await asyncio.sleep(0.5)
try:
await page.locator(LOGIN_BUTTON_SELECTOR).click(timeout=5000)
except Exception:
return ""
return await utils.find_login_qrcode(page, selector=QR_SELECTOR)
class AccountLoginSession:
"""一次针对**临时上下文**的扫码尝试。"""
def __init__(self, context: Any, page: Any) -> None:
self.status = STATUS_WAITING
self.message = "请用手机扫描二维码"
self.image = ""
self.started_at = time.time()
self.account: Optional[Dict[str, Any]] = None
self.cookie: str = ""
self.platform = PLATFORM_XHS
self._context = context
self._page = page
self._last_login_check = 0.0
@property
def elapsed(self) -> float:
return time.time() - self.started_at
async def refresh(self) -> None:
if self.status != STATUS_WAITING:
return
if self.elapsed > QR_TTL_SECONDS:
self.status = STATUS_EXPIRED
self.message = "二维码已超时,请重新获取"
return
# 前端每 2 秒问一次,但没必要每次都去打后台接口 —— 一次真实的网络往返
# 去确认一个通常还没发生的事件是浪费。
now = time.time()
if now - self._last_login_check < LOGIN_CHECK_INTERVAL_SECONDS:
return
self._last_login_check = now
try:
cookies = await self._context.cookies()
except Exception:
self.status = STATUS_ERROR
self.message = "登录窗口已被关闭,请重新获取"
return
cookie = "; ".join(f"{c['name']}={c['value']}" for c in cookies)
try:
info = await CreatorClient(cookie).fetch_user_info()
except CreatorApiError:
# 还没登录(或者刚扫、后端还没认),继续等。
return
if not info.get("user_id"):
return
# 登录成功:cookie 取自**这个临时上下文**,取完上下文就丢弃,
# 所以不会残留、也不会影响别的账号。
self.cookie = cookie
self.account = info
self.status = STATUS_SUCCESS
self.message = f"登录成功:{info.get('nickname') or info['user_id']}"
def snapshot(self) -> Dict[str, Any]:
return {
"status": self.status,
"message": self.message,
"image": self.image,
"elapsed": int(self.elapsed),
"expires_in": max(0, int(QR_TTL_SECONDS - self.elapsed)),
"account": self.account,
}
async def close(self) -> None:
"""关掉临时上下文。这是它存在的全部意义 —— 用完即弃。"""
for closer in (self._page.close, self._context.close):
try:
await closer()
except Exception:
pass
async def _teardown_locked() -> None:
global _current
if _current is not None:
await _current.close()
_current = None
async def start() -> Dict[str, Any]:
"""开一个临时上下文,打开登录页,取回二维码。"""
global _current, _last_result
async with _lock:
_last_result = None
await _teardown_locked()
try:
browser = await _connect()
except Exception as exc:
await _disconnect()
raise RuntimeError(
f"连接浏览器失败({_cdp_url()})。请确认服务器上的 Chrome 以 "
f"--remote-debugging-port 启动。原始错误:{exc}"
) from exc
# 临时上下文,不是 contexts[0]。这里刻意要一个干净的身份 ——
# 借用操作者自己的登录态会让"新增账号"变成"再读一遍当前账号"。
context = await browser.new_context()
page = await context.new_page()
try:
await page.goto(LOGIN_URL, wait_until="domcontentloaded", timeout=45000)
image = await _read_qr(page)
except Exception as exc:
try:
await page.close()
await context.close()
except Exception:
pass
raise RuntimeError(f"打开登录页失败:{exc}") from exc
session = AccountLoginSession(context, page)
session.image = image
if not image:
session.status = STATUS_ERROR
session.message = "页面上没找到二维码,请确认站点结构没有变化"
_current = session
return session.snapshot()
async def status() -> Dict[str, Any]:
async with _lock:
if _current is None:
if _last_result is not None:
return _last_result
return {
"status": STATUS_IDLE,
"message": "",
"image": "",
"elapsed": 0,
"expires_in": 0,
"account": None,
}
await _current.refresh()
return _current.snapshot()
async def remember_result(snapshot: Dict[str, Any]) -> None:
"""记住已完成的扫码结果,供后续轮询重复返回。"""
global _last_result
async with _lock:
_last_result = snapshot
async def take_cookie() -> Optional[str]:
"""取走已登录的 cookie 并结束会话。
由路由层在落库时调用。cookie 只经内存传递,**不进响应体** —— 它是凭证,
前端没有任何理由看到它。
"""
global _current
async with _lock:
if _current is None or _current.status != STATUS_SUCCESS:
return None
cookie = _current.cookie
await _teardown_locked()
return cookie
async def cancel() -> Dict[str, Any]:
global _last_result
async with _lock:
_last_result = None
await _teardown_locked()
return {
"status": STATUS_IDLE,
"message": "已取消",
"image": "",
"elapsed": 0,
"expires_in": 0,
"account": None,
}
async def shutdown() -> None:
async with _lock:
await _teardown_locked()
await _disconnect()