diff --git a/api/creator/__init__.py b/api/creator/__init__.py new file mode 100644 index 0000000..3f2dd12 --- /dev/null +++ b/api/creator/__init__.py @@ -0,0 +1,27 @@ +# -*- coding: utf-8 -*- +# Copyright (c) 2025 relakkes@gmail.com +# +# This file is part of MediaCrawler project. +# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/creator/__init__.py +# GitHub: https://github.com/NanmiCoder +# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 +# +# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: +# 1. 不得用于任何商业用途。 +# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 +# 3. 不得进行大规模爬取或对平台造成运营干扰。 +# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 +# 5. 不得用于任何非法或不当的用途。 +# +# 详细许可条款请参阅项目根目录下的LICENSE文件。 +# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 + +"""运营模块:管理自己的小红书账号,读取创作者后台的数据。 + +与 `api.monitor` 是**并列关系**,不是它的扩展。两者数据形状不同:监控是「每轮 +快照 + 差分」的公开互动数据,这里是创作者后台按日期给出的曝光/观看/完播等运营 +指标。硬塞进同一个模型会同时污染两边。 + +路线是**纯请求**(无浏览器),依据见 tools/probe_creator_api.py 的 Phase 0 实测: +签名可自造、主站 cookie 即可认证、接口与参数已与真实页面对齐。 +""" diff --git a/api/creator/client.py b/api/creator/client.py new file mode 100644 index 0000000..340237e --- /dev/null +++ b/api/creator/client.py @@ -0,0 +1,345 @@ +# -*- coding: utf-8 -*- +# Copyright (c) 2025 relakkes@gmail.com +# +# This file is part of MediaCrawler project. +# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/creator/client.py +# GitHub: https://github.com/NanmiCoder +# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 +# +# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: +# 1. 不得用于任何商业用途。 +# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 +# 3. 不得进行大规模爬取或对平台造成运营干扰。 +# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 +# 5. 不得用于任何非法或不当的用途。 +# +# 详细许可条款请参阅项目根目录下的LICENSE文件。 +# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 + +"""小红书创作者后台的纯请求客户端。 + +无浏览器:签名在本地算(见 signing.py),请求走 httpx。 + +**关于字段名的谨慎**:Phase 0 抓到的响应里,列表接口因为账号权限未生效而返回空壳 +(`data.result` 只有 `{success, code, message}` 没有数据),所以**真实字段名尚未亲眼 +见过**。因此每个指标都写成**多别名匹配**,并且解析不出来时存 `None` 而不是 0 —— +0 是真实值,None 是"不知道",两者混淆会让报表说谎。 +""" + +import re +from typing import Any, Dict, List, Optional + +import httpx + +from .signing import sign_xyw, signed_api + +CREATOR_ORIGIN = "https://creator.xiaohongshu.com" +DATA_ANALYSIS_PAGE = f"{CREATOR_ORIGIN}/statistics/data-analysis" + +USER_INFO_PATH = "/api/galaxy/user/info" +PERMISSION_PATH = "/api/galaxy/creator/datacenter/permission/query" +NOTE_LIST_PATH = "/api/galaxy/creator/datacenter/note/analyze/list" + +USER_AGENT = ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36" +) + +# 签名被网关拒绝时返回的响应体。区分它和普通业务错误很重要:406 说明签名写错了, +# 而应用层的 code/-100 说明签名没问题、只是没有登录态。 +SIGNATURE_REJECTED_MARKERS = ("code", -1) + + +class CreatorApiError(RuntimeError): + """调用创作者后台失败。``status`` 用于区分是网关拒绝还是业务错误。""" + + def __init__(self, message: str, status: int = 0, payload: Any = None): + super().__init__(message) + self.status = status + self.payload = payload + + +def trans_cookies(cookie_str: str) -> Dict[str, str]: + """把 cookie 字符串解析成字典。容忍末尾分号、换行和零散空格。""" + jar: Dict[str, str] = {} + for chunk in re.split(r"[;\n]", cookie_str or ""): + chunk = chunk.strip() + if not chunk or "=" not in chunk: + continue + name, value = chunk.split("=", 1) + name = name.strip() + if name: + jar[name] = value.strip() + return jar + + +def _pick(item: Dict[str, Any], *names: str) -> Any: + """按别名顺序取第一个存在的键。 + + 字段名来自二手资料,未亲眼验证,所以不赌单一命名。 + """ + for name in names: + if name in item and item[name] is not None: + return item[name] + return None + + +_COUNT_UNITS = {"万": 10_000, "w": 10_000, "W": 10_000, "亿": 100_000_000, "k": 1_000, "K": 1_000} + + +def as_int(value: Any) -> Optional[int]: + """解析计数。处理 "1.2万"、"1,234"、"123" 与已经是数字的情况。""" + if value is None or isinstance(value, bool): + return None + if isinstance(value, (int, float)): + return int(value) + + text = str(value).strip().replace(",", "").replace(" ", "") + if not text or text in ("-", "--", "暂无"): + return None + + unit = 1 + suffix = text[-1] + if suffix in _COUNT_UNITS: + unit = _COUNT_UNITS[suffix] + text = text[:-1] + try: + return int(float(text) * unit) + except ValueError: + return None + + +def as_float(value: Any) -> Optional[float]: + """解析比率。``"12.3%"`` -> 12.3;``"0.123"`` 原样返回数字。""" + if value is None or isinstance(value, bool): + return None + if isinstance(value, (int, float)): + return float(value) + + text = str(value).strip().replace("%", "") + try: + return float(text) + except ValueError: + return None + + +_DURATION_RE = re.compile(r"(?:(\d+)\s*分)?\s*(?:(\d+)\s*秒)?") + + +def as_seconds(value: Any) -> Optional[float]: + """解析时长。处理 ``"1分30秒"``、``"01:30"``、``"45"``(秒)。""" + if value is None or isinstance(value, bool): + return None + if isinstance(value, (int, float)): + return float(value) + + text = str(value).strip() + if not text or text in ("-", "--"): + return None + + if ":" in text: + parts = text.split(":") + try: + total = 0.0 + for part in parts: + total = total * 60 + float(part) + return total + except ValueError: + return None + + if "分" in text or "秒" in text: + minutes = re.search(r"(\d+)\s*分", text) + seconds = re.search(r"(\d+)\s*秒", text) + if not minutes and not seconds: + return None + return float(minutes.group(1) if minutes else 0) * 60 + float( + seconds.group(1) if seconds else 0 + ) + + try: + return float(text) + except ValueError: + return None + + +# 指标 -> 候选原始字段名。别名来自公开资料,权威与否只能等真实响应来验证。 +_FIELD_ALIASES: Dict[str, tuple] = { + "exposure": ("exposure", "exposure_count", "imp", "impression", "impression_count"), + "views": ("views", "view", "view_count", "watch", "watch_count", "read_count"), + "likes": ("likes", "like", "like_count", "liked_count"), + "comments": ("comments", "comment", "comment_count", "comments_count"), + "favorites": ("favorites", "favorite", "favorite_count", "collect", "collect_count", "collected_count"), + "shares": ("shares", "share", "share_count", "shared_count"), + "new_followers": ("new_followers", "fans_growth", "follower_growth", "increase_fans"), + "danmaku": ("danmaku", "danmaku_count", "barrage"), + "cover_ctr": ("cover_ctr", "cover_click_rate", "cover_click_ratio", "ctr"), + "avg_watch_seconds": ("avg_watch_seconds", "avg_watch_time", "average_watch_time"), + "two_second_exit_rate": ("two_second_exit_rate", "2s_exit_rate", "exit_rate_2s"), + "completion_rate": ("completion_rate", "finish_rate", "complete_rate"), +} + +_COUNT_FIELDS = { + "exposure", "views", "likes", "comments", "favorites", "shares", "new_followers", "danmaku", +} +_DURATION_FIELDS = {"avg_watch_seconds"} +_RATE_FIELDS = {"two_second_exit_rate", "completion_rate", "cover_ctr"} + +NOTE_ID_ALIASES = ("note_id", "noteId", "id", "content_id", "item_id") +TITLE_ALIASES = ("title", "content", "note_title", "display_title") +PUBLISH_TIME_ALIASES = ("publish_time", "publishTime", "post_time", "create_time", "time") + + +def _normalize_metric(name: str, raw: Any) -> Any: + if name in _COUNT_FIELDS: + return as_int(raw) + if name in _DURATION_FIELDS: + return as_seconds(raw) + if name in _RATE_FIELDS: + return as_float(raw) + return raw + + +def normalize_note(item: Dict[str, Any]) -> Dict[str, Any]: + """把一条原始记录规范化成落库用的字段。""" + note: Dict[str, Any] = { + "note_id": str(_pick(item, *NOTE_ID_ALIASES) or ""), + "title": str(_pick(item, *TITLE_ALIASES) or ""), + "publish_time": as_int(_pick(item, *PUBLISH_TIME_ALIASES)), + } + for name, aliases in _FIELD_ALIASES.items(): + note[name] = _normalize_metric(name, _pick(item, *aliases)) + return note + + +def find_note_list(payload: Any) -> List[Dict[str, Any]]: + """在响应里找出笔记数组。 + + 接口的**确切结构还没亲眼见过**(权限未生效时 `data.result` 里没有数据), + 所以不写死路径:遍历 JSON,挑出"看起来像一批笔记记录"的那个列表 —— + 元素是 dict,且至少带一个指标字段。找不到就返回空列表,让上层如实报"没数据"。 + """ + best: List[Dict[str, Any]] = [] + metric_keys = {alias for aliases in _FIELD_ALIASES.values() for alias in aliases} + + def walk(value: Any) -> None: + nonlocal best + if isinstance(value, dict): + for child in value.values(): + walk(child) + elif isinstance(value, list): + if value and isinstance(value[0], dict): + keys = set(value[0].keys()) + if keys & metric_keys and len(value) > len(best): + best = value + for child in value: + walk(child) + + walk(payload) + return best + + +class CreatorClient: + """一个账号的客户端。``cookie`` 就是它的全部身份。""" + + def __init__(self, cookie: str, timeout: float = 25.0): + self.cookies = trans_cookies(cookie) + self.a1 = self.cookies.get("a1", "") + self._timeout = timeout + + @property + def looks_authenticated(self) -> bool: + """签名需要 a1;没有它连请求都签不出来。""" + return bool(self.a1) + + def _headers(self, api: str, body: dict | None = None) -> Dict[str, str]: + cookie_header = "; ".join(f"{k}={v}" for k, v in self.cookies.items()) + return { + "user-agent": USER_AGENT, + "accept": "application/json, text/plain, */*", + "accept-language": "zh-CN,zh;q=0.9", + "origin": CREATOR_ORIGIN, + "referer": DATA_ANALYSIS_PAGE, + "cookie": cookie_header, + **sign_xyw(api, self.a1, body=body), + } + + async def _get(self, path: str, params: Dict[str, Any]) -> Dict[str, Any]: + if not self.looks_authenticated: + raise CreatorApiError("cookie 里没有 a1,无法完成签名", status=0) + + query = "&".join(f"{k}={v}" for k, v in params.items()) + api = signed_api(path, query) + url = f"{CREATOR_ORIGIN}{path}?{query}" if query else f"{CREATOR_ORIGIN}{path}" + + async with httpx.AsyncClient(timeout=self._timeout, follow_redirects=False) as client: + response = await client.get(url, headers=self._headers(api)) + return self._unwrap(response) + + @staticmethod + def _unwrap(response: httpx.Response) -> Dict[str, Any]: + if response.status_code == 406: + raise CreatorApiError( + "签名被网关拒绝(406)—— 待签字符串的拼法不对", status=406 + ) + try: + payload = response.json() + except Exception as exc: # noqa: BLE001 + raise CreatorApiError( + f"响应不是 JSON(HTTP {response.status_code})", status=response.status_code + ) from exc + + if response.status_code == 401 or payload.get("code") == -100: + raise CreatorApiError("登录态无效或已过期", status=401, payload=payload) + if not payload.get("success", True): + raise CreatorApiError( + str(payload.get("msg") or "接口返回失败"), + status=response.status_code, + payload=payload, + ) + return payload + + async def fetch_user_info(self) -> Dict[str, Any]: + """当前 cookie 属于哪个账号。登录后用它取名与去重。""" + payload = await self._get(USER_INFO_PATH, {}) + data = payload.get("data") or {} + return { + "user_id": str(data.get("userId") or ""), + "nickname": str(data.get("userName") or ""), + "avatar": str(data.get("userAvatar") or ""), + "red_id": str(data.get("redId") or ""), + "role": str(data.get("role") or ""), + "permissions": list(data.get("permissions") or []), + } + + async def fetch_permission(self) -> Dict[str, Any]: + """数据权限状态。 + + ``tip_msg`` 是后台原话(实测是"已为您申请数据权限,次日可查看"),照抄不改写 —— + 这条信息必须原样交给用户,它解释了"为什么没有数据"。 + """ + payload = await self._get(PERMISSION_PATH, {}) + data = payload.get("data") or {} + return { + "display": data.get("display"), + "status": data.get("status"), + "tip": str(data.get("tip_msg") or ""), + } + + async def fetch_note_list( + self, start_ms: int, end_ms: int, page_num: int = 1, page_size: int = 10, note_type: int = 0 + ) -> List[Dict[str, Any]]: + """按发布时间区间取笔记列表。 + + 参数与顺序**照抄真实页面的请求**(见 Phase 0 抓包),不要凭感觉改。 + """ + payload = await self._get( + NOTE_LIST_PATH, + { + "post_begin_time": start_ms, + "post_end_time": end_ms, + "type": note_type, + "page_size": page_size, + "page_num": page_num, + }, + ) + return [normalize_note(item) for item in find_note_list(payload)] diff --git a/api/creator/login.py b/api/creator/login.py new file mode 100644 index 0000000..913c775 --- /dev/null +++ b/api/creator/login.py @@ -0,0 +1,280 @@ +# -*- coding: utf-8 -*- +# Copyright (c) 2025 relakkes@gmail.com +# +# This file is part of MediaCrawler project. +# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/creator/login.py +# GitHub: https://github.com/NanmiCoder +# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 +# +# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: +# 1. 不得用于任何商业用途。 +# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 +# 3. 不得进行大规模爬取或对平台造成运营干扰。 +# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 +# 5. 不得用于任何非法或不当的用途。 +# +# 详细许可条款请参阅项目根目录下的LICENSE文件。 +# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 + +"""运营账号的扫码登录。 + +**与监控的扫码登录(`api.monitor.qrlogin`)有一处决定性差异**:那边把登录态写进 +浏览器**默认 profile**,因为爬虫要复用它;这边要的是 **cookie 字符串**,因为采集 +走纯 HTTP。所以这里每次登录都开一个**临时上下文**,扫完取出 cookie 就丢弃 —— + +* 登第二个账号不会把第一个顶掉(默认 profile 只能装一个登录态); +* 完全不影响监控那个登录态; +* 十个账号互不干扰。 + +扫码入口仍是主站(`www.xiaohongshu.com`):Phase 0 实测证明**主站的 cookie 就能 +认证创作者后台**,不需要单独的创作者登录。 +""" + +import asyncio +import json +import time +from typing import Any, Dict, Optional + +from playwright.async_api import async_playwright +from tools import utils + +from ..monitor.platforms import PLATFORM_XHS + +QR_TTL_SECONDS = 300 + +STATUS_IDLE = "idle" +STATUS_WAITING = "waiting" +STATUS_SUCCESS = "success" +STATUS_EXPIRED = "expired" +STATUS_ERROR = "error" + +LOGIN_URL = "https://www.xiaohongshu.com" +QR_SELECTOR = "xpath=//img[@class='qrcode-img']" +LOGIN_BUTTON_SELECTOR = "xpath=//*[@id='app']/div[1]/div[2]/div[1]/ul/div[1]/button" + +# 与 api/monitor/qrlogin.py 同一套判据:主站在 window.__INITIAL_STATE__ 里报告登录态, +# 而 user.loggedIn 是 Vue 的响应式引用,要 .value 解包才是布尔值。 +LOGIN_STATE_PROBE = """ +() => { + try { + const user = (window.__INITIAL_STATE__ || {}).user; + if (!user) return JSON.stringify({ known: false }); + let loggedIn = user.loggedIn; + if (loggedIn && typeof loggedIn === 'object' && 'value' in loggedIn) loggedIn = loggedIn.value; + let info = null; + try { info = user.userInfo || null; } catch (e) { info = null; } + const text = (v) => (v === null || v === undefined ? null : String(v)); + return JSON.stringify({ + known: true, + loggedIn: Boolean(loggedIn), + nickname: info ? text(info.nickname) : null + }); + } catch (e) { + return JSON.stringify({ known: false, error: String(e) }); + } +} +""" + +_lock = asyncio.Lock() +_current: Optional["AccountLoginSession"] = None + +_playwright: Any = None + + +def _cdp_url() -> str: + import os + + import config + + return os.getenv("MC_CDP_URL") or f"http://127.0.0.1:{config.CDP_DEBUG_PORT}" + + +async def _connect(): + global _playwright + if _playwright is None: + _playwright = await async_playwright().start() + return await _playwright.chromium.connect_over_cdp(_cdp_url(), timeout=15000) + + +async def _disconnect() -> None: + global _playwright + if _playwright is not None: + try: + await _playwright.stop() + except Exception: + pass + _playwright = None + + +async def _read_qr(page: Any) -> str: + image = await utils.find_login_qrcode(page, selector=QR_SELECTOR) + if image: + return image + # 登录框不一定自己弹出来,这是爬虫自身扫码流程的同款兜底。 + await asyncio.sleep(0.5) + try: + await page.locator(LOGIN_BUTTON_SELECTOR).click(timeout=5000) + except Exception: + return "" + return await utils.find_login_qrcode(page, selector=QR_SELECTOR) + + +class AccountLoginSession: + """一次针对**临时上下文**的扫码尝试。""" + + def __init__(self, context: Any, page: Any) -> None: + self.status = STATUS_WAITING + self.message = "请用手机扫描二维码" + self.image = "" + self.started_at = time.time() + self.account: Optional[Dict[str, Any]] = None + self.cookie: str = "" + self.platform = PLATFORM_XHS + self._context = context + self._page = page + + @property + def elapsed(self) -> float: + return time.time() - self.started_at + + async def refresh(self) -> None: + if self.status != STATUS_WAITING: + return + if self.elapsed > QR_TTL_SECONDS: + self.status = STATUS_EXPIRED + self.message = "二维码已超时,请重新获取" + return + + try: + raw = await self._page.evaluate(LOGIN_STATE_PROBE) + state = json.loads(raw) if isinstance(raw, str) else {} + except Exception: + self.status = STATUS_ERROR + self.message = "二维码所在页面已被关闭,请重新获取" + return + + if not state.get("logged_in"): + return + + # 登录成功:从**这个临时上下文**取 cookie。取完上下文就丢弃, + # 所以不会残留、也不会影响别的账号。 + cookies = await self._context.cookies() + self.cookie = "; ".join(f"{c['name']}={c['value']}" for c in cookies) + self.status = STATUS_SUCCESS + self.message = "已获取登录态,正在识别账号…" + + def snapshot(self) -> Dict[str, Any]: + return { + "status": self.status, + "message": self.message, + "image": self.image, + "elapsed": int(self.elapsed), + "expires_in": max(0, int(QR_TTL_SECONDS - self.elapsed)), + "account": self.account, + } + + async def close(self) -> None: + """关掉临时上下文。这是它存在的全部意义 —— 用完即弃。""" + for closer in (self._page.close, self._context.close): + try: + await closer() + except Exception: + pass + + +async def _teardown_locked() -> None: + global _current + if _current is not None: + await _current.close() + _current = None + + +async def start() -> Dict[str, Any]: + """开一个临时上下文,打开登录页,取回二维码。""" + global _current + + async with _lock: + await _teardown_locked() + + try: + browser = await _connect() + except Exception as exc: + await _disconnect() + raise RuntimeError( + f"连接浏览器失败({_cdp_url()})。请确认服务器上的 Chrome 以 " + f"--remote-debugging-port 启动。原始错误:{exc}" + ) from exc + + # 临时上下文,不是 contexts[0]。这里刻意要一个干净的身份 —— + # 借用操作者自己的登录态会让"新增账号"变成"再读一遍当前账号"。 + context = await browser.new_context() + page = await context.new_page() + + try: + await page.goto(LOGIN_URL, wait_until="domcontentloaded", timeout=45000) + image = await _read_qr(page) + except Exception as exc: + try: + await page.close() + await context.close() + except Exception: + pass + raise RuntimeError(f"打开登录页失败:{exc}") from exc + + session = AccountLoginSession(context, page) + session.image = image + if not image: + session.status = STATUS_ERROR + session.message = "页面上没找到二维码,请确认站点结构没有变化" + + _current = session + return session.snapshot() + + +async def status() -> Dict[str, Any]: + async with _lock: + if _current is None: + return { + "status": STATUS_IDLE, + "message": "", + "image": "", + "elapsed": 0, + "expires_in": 0, + "account": None, + } + await _current.refresh() + return _current.snapshot() + + +async def take_cookie() -> Optional[str]: + """取走已登录的 cookie 并结束会话。 + + 由路由层在落库时调用。cookie 只经内存传递,**不进响应体** —— 它是凭证, + 前端没有任何理由看到它。 + """ + global _current + async with _lock: + if _current is None or _current.status != STATUS_SUCCESS: + return None + cookie = _current.cookie + await _teardown_locked() + return cookie + + +async def cancel() -> Dict[str, Any]: + async with _lock: + await _teardown_locked() + return { + "status": STATUS_IDLE, + "message": "已取消", + "image": "", + "elapsed": 0, + "expires_in": 0, + "account": None, + } + + +async def shutdown() -> None: + async with _lock: + await _teardown_locked() + await _disconnect() diff --git a/api/creator/models.py b/api/creator/models.py new file mode 100644 index 0000000..d131d7e --- /dev/null +++ b/api/creator/models.py @@ -0,0 +1,129 @@ +# -*- coding: utf-8 -*- +# Copyright (c) 2025 relakkes@gmail.com +# +# This file is part of MediaCrawler project. +# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/creator/models.py +# GitHub: https://github.com/NanmiCoder +# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 +# +# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: +# 1. 不得用于任何商业用途。 +# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 +# 3. 不得进行大规模爬取或对平台造成运营干扰。 +# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 +# 5. 不得用于任何非法或不当的用途。 +# +# 详细许可条款请参阅项目根目录下的LICENSE文件。 +# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 + +"""运营模块的数据模型。 + +**刻意复用 `MonitorBase`**:这样 `init_db` 的 `create_all` 会顺手建出新表,而 +`_ensure_columns`(已改为按模型元数据推导)也会自动给新表补字段 —— 不必再维护一份 +建表语句。表落在同一个库里,与监控互不干扰。 +""" + +from typing import Optional + +from sqlalchemy import BigInteger, Float, ForeignKey, Index, Integer, String, Text +from sqlalchemy.orm import Mapped, mapped_column, relationship + +from ..monitor.models import MonitorBase + +# 创作者后台的数据权限是「首次访问时自动申请、次日生效」。这个状态必须如实呈现: +# 显示成"没数据"会让人以为采集坏了,实际是在等审批。 +PERMISSION_UNKNOWN = "unknown" +PERMISSION_PENDING = "pending" # 已申请,未生效(提示语:"次日可查看") +PERMISSION_ACTIVE = "active" +PERMISSION_MISSING = "missing" # 接口明确说没有权限 + +# 账号自身的可用性。 +ACCOUNT_OK = "ok" +ACCOUNT_EXPIRED = "expired" # cookie 失效,需要重新扫码 +ACCOUNT_ERROR = "error" + + +class CreatorAccount(MonitorBase): + """一个自己的小红书账号。 + + 纯请求路线下,**一个账号的全部身份就是一份 cookie** —— 没有浏览器 profile、 + 没有独立目录。所以"多账号"在这里只是表里的多行,不是多套运行环境。 + + `cookie` 是凭证:与监控的 cookie 同样对待,只存库、绝不回显接口。 + """ + + __tablename__ = "creator_account" + + id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True) + # 展示名。优先用后台返回的昵称,用户可以改。 + nickname: Mapped[str] = mapped_column(String(128), nullable=False, default="") + # 创作者后台的账号标识,由 /api/galaxy/user/info 返回,用于去重。 + user_id: Mapped[str] = mapped_column(String(64), nullable=False, default="", index=True) + red_id: Mapped[str] = mapped_column(String(64), nullable=False, default="") + avatar: Mapped[str] = mapped_column(Text, nullable=False, default="") + + cookie: Mapped[str] = mapped_column(Text, nullable=False, default="") + + status: Mapped[str] = mapped_column(String(16), nullable=False, default=ACCOUNT_OK) + permission_status: Mapped[str] = mapped_column( + String(16), nullable=False, default=PERMISSION_UNKNOWN + ) + # 后台原话,例如"已为您申请数据权限,次日可查看"。照抄,不改写。 + permission_tip: Mapped[str] = mapped_column(Text, nullable=False, default="") + + last_checked_at: Mapped[Optional[int]] = mapped_column(BigInteger) + last_synced_at: Mapped[Optional[int]] = mapped_column(BigInteger) + last_error: Mapped[Optional[str]] = mapped_column(Text) + + created_at: Mapped[int] = mapped_column(BigInteger, nullable=False) + updated_at: Mapped[int] = mapped_column(BigInteger, nullable=False) + + notes: Mapped[list["CreatorNoteStat"]] = relationship( + back_populates="account", cascade="all, delete-orphan" + ) + + +class CreatorNoteStat(MonitorBase): + """一篇作品在某个采集时点的运营数据。 + + 创作者后台给的是**累计值**(截至查询时点),所以反复采集天然形成时间序列 —— + 与监控的"快照 + 差分"是同一个思路,因此这里保留 `captured_at` 而不是覆盖写。 + """ + + __tablename__ = "creator_note_stat" + + id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True) + account_id: Mapped[int] = mapped_column( + ForeignKey("creator_account.id", ondelete="CASCADE"), nullable=False, index=True + ) + note_id: Mapped[str] = mapped_column(String(64), nullable=False, index=True) + title: Mapped[str] = mapped_column(Text, nullable=False, default="") + # 发布时间(毫秒)。后台按发布时间筛选,这是它的主时间轴。 + publish_time: Mapped[Optional[int]] = mapped_column(BigInteger) + + # --- 运营指标 --------------------------------------------------------- + # 计数用 BigInteger:曝光量可以很大,用 INT 迟早溢出。 + exposure: Mapped[Optional[int]] = mapped_column(BigInteger) + views: Mapped[Optional[int]] = mapped_column(BigInteger) + likes: Mapped[Optional[int]] = mapped_column(BigInteger) + comments: Mapped[Optional[int]] = mapped_column(BigInteger) + favorites: Mapped[Optional[int]] = mapped_column(BigInteger) + shares: Mapped[Optional[int]] = mapped_column(BigInteger) + new_followers: Mapped[Optional[int]] = mapped_column(BigInteger) + danmaku: Mapped[Optional[int]] = mapped_column(BigInteger) + + # 比率与时长。后台返回的可能是 "12.3%"/"1分30秒" 这类字符串,解析不了的存 NULL + # 而不是 0 —— 与监控层的口径一致:0 是真实值,NULL 是"不知道"。 + cover_ctr: Mapped[Optional[float]] = mapped_column(Float) + avg_watch_seconds: Mapped[Optional[float]] = mapped_column(Float) + two_second_exit_rate: Mapped[Optional[float]] = mapped_column(Float) + completion_rate: Mapped[Optional[float]] = mapped_column(Float) + + captured_at: Mapped[int] = mapped_column(BigInteger, nullable=False, index=True) + + account: Mapped["CreatorAccount"] = relationship(back_populates="notes") + + __table_args__ = ( + # 同一个时点同一篇只留一行,重复同步不会堆积。 + Index("ix_creator_note_stat_unique", "account_id", "note_id", "captured_at", unique=True), + ) diff --git a/api/creator/service.py b/api/creator/service.py new file mode 100644 index 0000000..59b42e0 --- /dev/null +++ b/api/creator/service.py @@ -0,0 +1,354 @@ +# -*- coding: utf-8 -*- +# Copyright (c) 2025 relakkes@gmail.com +# +# This file is part of MediaCrawler project. +# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/creator/service.py +# GitHub: https://github.com/NanmiCoder +# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 +# +# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: +# 1. 不得用于任何商业用途。 +# 2. 使用时应遵守对应平台的使用条款和robots.txt规则。 +# 3. 不得进行大规模爬取或对平台造成运营干扰。 +# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 +# 5. 不得用于任何非法或不当的用途。 +# +# 详细许可条款请参阅项目根目录下的LICENSE文件。 +# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 + +"""运营账号的增删查与数据同步。 + +一条贯穿全文件的规则:**cookie 是凭证,永远不出现在返回给上层的结构里。** +对外只给 `has_cookie` 这样的布尔量,与监控层对 cookie 的处理保持一致。 +""" + +import asyncio +from datetime import datetime, time, timedelta +from typing import Any, Dict, List, Optional + +from sqlalchemy import delete, func, select +from sqlalchemy.ext.asyncio import AsyncSession + +from tools.time_util import get_current_timestamp + +from .client import CreatorApiError, CreatorClient +from .models import ( + ACCOUNT_ERROR, + ACCOUNT_EXPIRED, + ACCOUNT_OK, + PERMISSION_ACTIVE, + PERMISSION_MISSING, + PERMISSION_PENDING, + PERMISSION_UNKNOWN, + CreatorAccount, + CreatorNoteStat, +) + +# 同步一次最多翻多少页。后台默认一页 10 条,200 页足以覆盖任何正常账号, +# 同时防止"接口不返回 has_more"时无限翻下去。 +MAX_SYNC_PAGES = 200 +PAGE_SIZE = 10 + + +def _account_dict(account: CreatorAccount, note_count: int = 0) -> Dict[str, Any]: + """账号的对外表示。**刻意不含 cookie。**""" + return { + "id": account.id, + "nickname": account.nickname, + "user_id": account.user_id, + "red_id": account.red_id, + "avatar": account.avatar, + "status": account.status, + "permission_status": account.permission_status, + # 后台原话照抄。"次日可查看"这类信息只能由它自己说,改写就失真了。 + "permission_tip": account.permission_tip, + "last_checked_at": account.last_checked_at, + "last_synced_at": account.last_synced_at, + "last_error": account.last_error, + "has_cookie": bool(account.cookie), + "created_at": account.created_at, + "note_count": note_count, + } + + +async def list_accounts(session: AsyncSession) -> List[Dict[str, Any]]: + accounts = list( + (await session.scalars(select(CreatorAccount).order_by(CreatorAccount.id))).all() + ) + counts = dict( + ( + await session.execute( + select(CreatorNoteStat.account_id, func.count(func.distinct(CreatorNoteStat.note_id))) + .group_by(CreatorNoteStat.account_id) + ) + ).all() + ) + return [_account_dict(account, counts.get(account.id, 0)) for account in accounts] + + +async def get_account(session: AsyncSession, account_id: int) -> CreatorAccount: + account = await session.get(CreatorAccount, account_id) + if account is None: + raise ValueError(f"账号 {account_id} 不存在") + return account + + +async def account_detail(session: AsyncSession, account_id: int) -> Dict[str, Any]: + account = await get_account(session, account_id) + notes = await latest_notes(session, account_id) + return { + "account": _account_dict(account, len(notes)), + "notes": notes, + "summary": _summarize(notes), + } + + +def _summarize(notes: List[Dict[str, Any]]) -> Dict[str, Any]: + """账号级汇总。取最后一轮快照的累计值之和。""" + totals = { + key: 0 + for key in ("exposure", "views", "likes", "comments", "favorites", "shares", "new_followers") + } + for note in notes: + for key in totals: + value = note.get(key) + if isinstance(value, (int, float)): + totals[key] += int(value) + return totals + + +async def latest_notes(session: AsyncSession, account_id: int) -> List[Dict[str, Any]]: + """每个作品取**最近一次**快照。 + + 表里保留全部历史(换个时点就是一条新行),但列表只该展示"现在",否则同一个 + 作品会在列表里出现多次。 + """ + newest = ( + select( + CreatorNoteStat.note_id, + func.max(CreatorNoteStat.captured_at).label("captured_at"), + ) + .where(CreatorNoteStat.account_id == account_id) + .group_by(CreatorNoteStat.note_id) + .subquery() + ) + rows = ( + await session.scalars( + select(CreatorNoteStat) + .join( + newest, + (CreatorNoteStat.note_id == newest.c.note_id) + & (CreatorNoteStat.captured_at == newest.c.captured_at), + ) + .where(CreatorNoteStat.account_id == account_id) + .order_by(CreatorNoteStat.publish_time.desc().nullslast()) + ) + ).all() + return [_note_dict(row) for row in rows] + + +def _note_dict(row: CreatorNoteStat) -> Dict[str, Any]: + return { + "note_id": row.note_id, + "title": row.title, + "publish_time": row.publish_time, + "exposure": row.exposure, + "views": row.views, + "likes": row.likes, + "comments": row.comments, + "favorites": row.favorites, + "shares": row.shares, + "new_followers": row.new_followers, + "danmaku": row.danmaku, + "cover_ctr": row.cover_ctr, + "avg_watch_seconds": row.avg_watch_seconds, + "two_second_exit_rate": row.two_second_exit_rate, + "completion_rate": row.completion_rate, + "captured_at": row.captured_at, + } + + +async def upsert_account_from_cookie(session: AsyncSession, cookie: str) -> Dict[str, Any]: + """用一份 cookie 识别并保存账号。 + + 识别靠 `user/info` 而不是让用户填名字 —— 填错名字只会让后面所有数据对不上号。 + 已有同 `user_id` 的账号则更新它的 cookie(重新登录)。 + """ + client = CreatorClient(cookie) + if not client.looks_authenticated: + raise ValueError("这份 cookie 里没有 a1,无法签名,请重新扫码") + + try: + info = await client.fetch_user_info() + except CreatorApiError as exc: + raise ValueError(f"登录态无法使用:{exc}") from exc + + if not info.get("user_id"): + raise ValueError("接口没有返回账号标识,可能登录态无效") + + now = get_current_timestamp() + account = await session.scalar( + select(CreatorAccount).where(CreatorAccount.user_id == info["user_id"]) + ) + if account is None: + account = CreatorAccount(created_at=now) + session.add(account) + + account.nickname = info.get("nickname") or account.nickname or "未命名账号" + account.user_id = info["user_id"] + account.red_id = info.get("red_id") or "" + account.avatar = info.get("avatar") or "" + account.cookie = cookie + account.status = ACCOUNT_OK + account.last_error = None + account.last_checked_at = now + account.updated_at = now + + await session.flush() + + # 顺手把权限状态也拉一次:新账号几乎必然处于"已申请、次日生效", + # 当场告诉用户,比让他明天再回来问要好。 + await refresh_permission(session, account) + + return _account_dict(account) + + +async def refresh_permission(session: AsyncSession, account: CreatorAccount) -> None: + """查询并记录数据权限状态。失败不影响账号本身可用。""" + try: + permission = await CreatorClient(account.cookie).fetch_permission() + except CreatorApiError as exc: + if exc.status == 401: + account.status = ACCOUNT_EXPIRED + account.last_error = str(exc) + else: + account.last_error = str(exc) + account.updated_at = get_current_timestamp() + return + + display = permission.get("display") + status = permission.get("status") + account.permission_tip = permission.get("tip") or "" + + if display or status: + account.permission_status = PERMISSION_ACTIVE + elif account.permission_tip: + # 有提示语但未开通 —— 实测就是"已为您申请数据权限,次日可查看"。 + account.permission_status = PERMISSION_PENDING + else: + account.permission_status = PERMISSION_MISSING + + account.status = ACCOUNT_OK + account.last_error = None + account.last_checked_at = get_current_timestamp() + account.updated_at = account.last_checked_at + + +async def check_account(session: AsyncSession, account_id: int) -> Dict[str, Any]: + """重新检测一个账号:登录态还在不在、权限开通没有。""" + account = await get_account(session, account_id) + await refresh_permission(session, account) + count = ( + await session.execute( + select(func.count(func.distinct(CreatorNoteStat.note_id))).where( + CreatorNoteStat.account_id == account_id + ) + ) + ).scalar() or 0 + return _account_dict(account, count) + + +async def delete_account(session: AsyncSession, account_id: int) -> None: + account = await get_account(session, account_id) + await session.execute( + delete(CreatorNoteStat).where(CreatorNoteStat.account_id == account_id) + ) + await session.delete(account) + + +def _day_bounds(days: int) -> tuple[int, int]: + """最近 N 天的起止(毫秒)。与后台的按发布时间筛选对齐。""" + today = datetime.now() + end = int(datetime.combine(today.date(), time(23, 59, 59)).timestamp() * 1000) + start = int( + datetime.combine((today - timedelta(days=days)).date(), time(0, 0, 0)).timestamp() * 1000 + ) + return start, end + + +async def sync_account( + session: AsyncSession, account_id: int, days: int = 90 +) -> Dict[str, Any]: + """拉取一个账号的作品运营数据并落库。 + + 权限未生效时接口返回的是**空壳成功**(`data.result` 里没有数据),不是错误 —— + 所以"同步成功但 0 条"是正常结果,必须如实回报,不能让用户以为采集坏了。 + """ + account = await get_account(session, account_id) + if not account.cookie: + raise ValueError("该账号没有可用的登录态,请重新扫码") + + client = CreatorClient(account.cookie) + start_ms, end_ms = _day_bounds(days) + now = get_current_timestamp() + + collected: List[Dict[str, Any]] = [] + for page in range(1, MAX_SYNC_PAGES + 1): + try: + batch = await client.fetch_note_list(start_ms, end_ms, page_num=page, page_size=PAGE_SIZE) + except CreatorApiError as exc: + account.last_error = str(exc) + if exc.status == 401: + account.status = ACCOUNT_EXPIRED + account.updated_at = get_current_timestamp() + raise ValueError(f"同步失败:{exc}") from exc + + collected.extend(batch) + if len(batch) < PAGE_SIZE: + break + await asyncio.sleep(0.6) # 对后台客气一点,这是自己的账号但仍是自动化访问 + + # 先删掉本时点可能存在的重复行,再写入 —— 表上有 (account, note, captured_at) + # 唯一索引,重复同步不该报错。 + await session.execute( + delete(CreatorNoteStat).where( + CreatorNoteStat.account_id == account_id, CreatorNoteStat.captured_at == now + ) + ) + for note in collected: + if not note.get("note_id"): + continue + session.add( + CreatorNoteStat( + account_id=account_id, + note_id=note["note_id"], + title=note.get("title") or "", + publish_time=note.get("publish_time"), + exposure=note.get("exposure"), + views=note.get("views"), + likes=note.get("likes"), + comments=note.get("comments"), + favorites=note.get("favorites"), + shares=note.get("shares"), + new_followers=note.get("new_followers"), + danmaku=note.get("danmaku"), + cover_ctr=note.get("cover_ctr"), + avg_watch_seconds=note.get("avg_watch_seconds"), + two_second_exit_rate=note.get("two_second_exit_rate"), + completion_rate=note.get("completion_rate"), + captured_at=now, + ) + ) + + account.last_synced_at = now + account.last_error = None + account.updated_at = now + await refresh_permission(session, account) + await session.flush() + + return { + "account_id": account_id, + "fetched": len(collected), + "permission_status": account.permission_status, + "permission_tip": account.permission_tip, + } diff --git a/api/creator/signing.py b/api/creator/signing.py new file mode 100644 index 0000000..290b063 --- /dev/null +++ b/api/creator/signing.py @@ -0,0 +1,107 @@ +# -*- coding: utf-8 -*- +# Copyright (c) 2025 relakkes@gmail.com +# +# This file is part of MediaCrawler project. +# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/creator/signing.py +# GitHub: https://github.com/NanmiCoder +# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 +# +# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: +# 1. 不得用于任何商业用途。 +# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 +# 3. 不得进行大规模爬取或对平台造成运营干扰。 +# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 +# 5. 不得用于任何非法或不当的用途。 +# +# 详细许可条款请参阅项目根目录下的LICENSE文件。 +# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 + +"""创作者后台的请求签名(XYW_ 方案)。 + +主站与创作者后台用的是**两套不同的签名**:主站是 VMP 的 `XYS_`,创作者后台是 +`XYW_`。后者简单得多 —— MD5 → base64 → AES-128-CBC,密钥与 IV 都是硬编码常量, +纯 Python 可算,不需要浏览器。 + +常量与 `xhshow/config/config.py` 逐字节一致(该库也据此实现了 `sign_xyw`), +并与独立的逆向实现 xiaohongshu-cli/creator_signing.py 互相印证。 + +**三条实测结论**(tools/probe_creator_api.py 的 Phase 0 输出): +1. 待签字符串必须是 `url=` + 路径 + 查询串 的形式。只给路径、或去掉 `url=` 前缀, + 网关一律返回 **406**;写法正确时签名通过。 +2. `appId` 用 `ugc`(创作者平台的取值),不是主站的 `xhs-pc-web`。 +3. 不带 cookie 时返回的是应用层的 401「无登录信息」而非 406 —— 说明签名每次都过了, + 认证是独立的一层。 +""" + +import base64 +import hashlib +import json +from datetime import datetime + +XYW_AES_KEY = b"7cc4adla5ay0701v" +XYW_AES_IV = b"4uzjr7mbsibcaldp" + +# 与 xhshow 的 XYW_ENV_FLAGS_DEFAULT 一致。含义未知,但改了签名就不被接受。 +XYW_ENV_FLAGS = "0|0|0|1|0|0|1|0|0|0|1|0|0|0|0|1|0|0|0" + +XYW_PREFIX = "XYW_" +XYW_SIGN_SVN = "56" +XYW_SIGN_TYPE = "x2" +XYW_SIGN_VERSION = "1" + +# 创作者平台的 appId。用主站的 xhs-pc-web 会被拒。 +CREATOR_APP_ID = "ugc" + + +def _aes_encrypt_hex(plaintext: str) -> str: + from Crypto.Cipher import AES + from Crypto.Util.Padding import pad + + cipher = AES.new(XYW_AES_KEY, AES.MODE_CBC, XYW_AES_IV) + return cipher.encrypt(pad(plaintext.encode("utf-8"), AES.block_size)).hex() + + +def sign_xyw( + api: str, + a1: str, + app_id: str = CREATOR_APP_ID, + body: dict | None = None, + timestamp_ms: int | None = None, +) -> dict[str, str]: + """为一次创作者后台请求生成 ``x-s`` / ``x-t`` 请求头。 + + ``api`` 必须是待签的完整字符串:``url=`` 加路径,GET 请求还要带上查询串。 + POST 的 JSON body 追加在其后(紧凑分隔符、不转义非 ASCII),与参考实现一致。 + """ + content = api + if body is not None: + content += json.dumps(body, separators=(",", ":"), ensure_ascii=False) + + if timestamp_ms is None: + timestamp_ms = int(datetime.now().timestamp() * 1000) + + digest = hashlib.md5(content.encode("utf-8")).hexdigest() + plaintext = f"x1={digest};x2={XYW_ENV_FLAGS};x3={a1};x4={timestamp_ms};" + encoded = base64.b64encode(plaintext.encode("utf-8")).decode("utf-8") + + envelope = { + "signSvn": XYW_SIGN_SVN, + "signType": XYW_SIGN_TYPE, + "appId": app_id, + "signVersion": XYW_SIGN_VERSION, + "payload": _aes_encrypt_hex(encoded), + } + x_s = XYW_PREFIX + base64.b64encode( + json.dumps(envelope, separators=(",", ":")).encode("utf-8") + ).decode("utf-8") + + return {"x-s": x_s, "x-t": str(timestamp_ms)} + + +def signed_api(url_path: str, query: str = "") -> str: + """把路径与查询串拼成待签字符串。 + + 单独抽出来是因为这个格式**没有文档**,只能靠实测固定下来 —— 写错就是 406, + 而 406 的响应体 ``{"code":-1,"success":false}`` 完全看不出错在哪。 + """ + return f"url={url_path}?{query}" if query else f"url={url_path}" diff --git a/api/main.py b/api/main.py index f1854b0..1eb9656 100644 --- a/api/main.py +++ b/api/main.py @@ -48,6 +48,7 @@ from .auth import ensure_initial_credential, require_auth from .routers import ( auth_router, crawler_router, + creator_router, data_router, monitor_router, settings_router, @@ -64,6 +65,7 @@ async def lifespan(_app: FastAPI): browser session the way the log broadcaster is -- a scheduled run has to happen whether or not anyone has the UI open. """ + from .creator.login import shutdown as shutdown_creator_login from .monitor.db import dispose_engine, init_db from .monitor.qrlogin import shutdown as shutdown_qrlogin from .monitor.scheduler import monitor_scheduler @@ -103,6 +105,9 @@ async def lifespan(_app: FastAPI): # Drops the tab a QR login may have opened and stops the Playwright # client; leaving them would strand a driver process on every restart. await shutdown_qrlogin() + # Same for the operator's account logins, which run in throwaway browser + # contexts -- those would otherwise be left open in the operator's Chrome. + await shutdown_creator_login() await dispose_engine() @@ -150,6 +155,7 @@ app.add_middleware( # more importantly, never sees WebSocket scopes at all. app.include_router(auth_router, prefix="/api") app.include_router(crawler_router, prefix="/api", dependencies=[Depends(require_auth)]) +app.include_router(creator_router, prefix="/api", dependencies=[Depends(require_auth)]) app.include_router(data_router, prefix="/api", dependencies=[Depends(require_auth)]) app.include_router(monitor_router, prefix="/api", dependencies=[Depends(require_auth)]) app.include_router(settings_router, prefix="/api", dependencies=[Depends(require_auth)]) diff --git a/api/routers/__init__.py b/api/routers/__init__.py index 70c6dc6..0531baa 100644 --- a/api/routers/__init__.py +++ b/api/routers/__init__.py @@ -18,6 +18,7 @@ from .auth import router as auth_router from .crawler import router as crawler_router +from .creator import router as creator_router from .data import router as data_router from .monitor import router as monitor_router from .settings import router as settings_router @@ -26,6 +27,7 @@ from .websocket import router as websocket_router __all__ = [ "auth_router", "crawler_router", + "creator_router", "data_router", "monitor_router", "settings_router", diff --git a/api/routers/creator.py b/api/routers/creator.py new file mode 100644 index 0000000..fd0aa1c --- /dev/null +++ b/api/routers/creator.py @@ -0,0 +1,146 @@ +# -*- coding: utf-8 -*- +# Copyright (c) 2025 relakkes@gmail.com +# +# This file is part of MediaCrawler project. +# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/routers/creator.py +# GitHub: https://github.com/NanmiCoder +# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 +# +# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: +# 1. 不得用于任何商业用途。 +# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 +# 3. 不得进行大规模爬取或对平台造成运营干扰。 +# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 +# 5. 不得用于任何非法或不当的用途。 +# +# 详细许可条款请参阅项目根目录下的LICENSE文件。 +# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 + +"""运营模块的 HTTP 接口。""" + +import asyncio +from typing import Set + +from fastapi import APIRouter, HTTPException, Query + +from ..creator import login as creator_login +from ..creator import service +from ..monitor.db import get_session + +router = APIRouter(prefix="/creator", tags=["creator"]) + +# 后台同步任务要留强引用:asyncio 只持弱引用,否则任务可能在跑完前被回收。 +_sync_tasks: Set[asyncio.Task] = set() + + +def _bad_request(exc: ValueError) -> HTTPException: + return HTTPException(status_code=400, detail=str(exc)) + + +@router.get("/accounts") +async def list_accounts(): + """账号列表。**不含 cookie**,只给 ``has_cookie``。""" + async with get_session() as session: + return {"accounts": await service.list_accounts(session)} + + +@router.get("/accounts/{account_id}") +async def get_account_detail(account_id: int): + async with get_session() as session: + try: + return await service.account_detail(session, account_id) + except ValueError as exc: + raise HTTPException(status_code=404, detail=str(exc)) + + +@router.delete("/accounts/{account_id}") +async def delete_account(account_id: int): + async with get_session() as session: + try: + await service.delete_account(session, account_id) + except ValueError as exc: + raise HTTPException(status_code=404, detail=str(exc)) + return {"message": "账号已删除"} + + +@router.post("/accounts/{account_id}/check") +async def check_account(account_id: int): + """重测登录态与数据权限。""" + async with get_session() as session: + try: + return await service.check_account(session, account_id) + except ValueError as exc: + raise HTTPException(status_code=404, detail=str(exc)) + + +@router.post("/accounts/{account_id}/sync") +async def sync_account( + account_id: int, days: int = Query(default=90, ge=1, le=730) +): + """拉取作品数据。 + + **放后台跑**:要分页、还要按账号节流,几分钟很正常,而前端请求超时是 30 秒。 + 前端靠轮询账号列表里的 `last_synced_at` / `last_error` 看结果。 + """ + async with get_session() as session: + try: + await service.get_account(session, account_id) + except ValueError as exc: + raise HTTPException(status_code=404, detail=str(exc)) + + task = asyncio.create_task(_sync_in_background(account_id, days)) + _sync_tasks.add(task) + task.add_done_callback(_sync_tasks.discard) + return {"message": "同步已开始", "days": days} + + +async def _sync_in_background(account_id: int, days: int) -> None: + try: + async with get_session() as session: + result = await service.sync_account(session, account_id, days) + print(f"[creator] 账号 {account_id} 同步完成,取回 {result['fetched']} 条") + except Exception as exc: # noqa: BLE001 - 后台任务不能让异常逃逸成静默失败 + print(f"[creator] 账号 {account_id} 同步失败: {exc}") + + +# --------------------------------------------------------------------------- +# 扫码新增账号 +# --------------------------------------------------------------------------- +# +# 每次登录开一个**临时浏览器上下文**,扫完取出 cookie 就丢弃 —— 这样登第二个账号 +# 不会把第一个顶掉,也不影响监控那个登录态。cookie 只在内存里从 login 模块传到 +# 这里落库,**不进响应体**。 + + +@router.post("/login") +async def start_login(): + try: + return await creator_login.start() + except RuntimeError as exc: + raise HTTPException(status_code=502, detail=str(exc)) + + +@router.get("/login") +async def poll_login(): + """轮询扫码结果;一旦成功就把账号落库并返回它。""" + snapshot = await creator_login.status() + + if snapshot["status"] == creator_login.STATUS_SUCCESS and snapshot.get("account") is None: + cookie = await creator_login.take_cookie() + if cookie: + try: + async with get_session() as session: + account = await service.upsert_account_from_cookie(session, cookie) + except ValueError as exc: + snapshot["status"] = creator_login.STATUS_ERROR + snapshot["message"] = f"扫码成功但保存账号失败:{exc}" + return snapshot + snapshot["account"] = account + snapshot["message"] = f"已添加账号:{account['nickname']}" + + return snapshot + + +@router.delete("/login") +async def cancel_login(): + return await creator_login.cancel() diff --git a/tests/test_creator_client.py b/tests/test_creator_client.py new file mode 100644 index 0000000..02b6157 --- /dev/null +++ b/tests/test_creator_client.py @@ -0,0 +1,241 @@ +# -*- coding: utf-8 -*- +"""创作者后台客户端的解析与签名。 + +**字段名尚未亲眼验证过**:Phase 0 抓响应时账号的数据权限还没生效,列表接口返回的是 +空壳(`data.result` 里只有 `{success, code, message}`)。所以这些解析写成多别名匹配, +而这份测试就是它的规格 —— 等真实响应到手,先跑这里看哪些假设破了。 +""" + +import pytest + +from api.creator import signing +from api.creator.client import ( + CreatorClient, + as_float, + as_int, + as_seconds, + find_note_list, + normalize_note, + trans_cookies, +) +from api.creator.models import CreatorAccount +from api.creator.service import _account_dict + + +# --- cookie 解析 ----------------------------------------------------------- + + +@pytest.mark.parametrize( + "raw, expected", + [ + ("a1=abc; web_session=xyz", {"a1": "abc", "web_session": "xyz"}), + ("a1=abc;web_session=xyz;", {"a1": "abc", "web_session": "xyz"}), + ("a1=abc\nweb_session=xyz", {"a1": "abc", "web_session": "xyz"}), + (" a1 = abc ; ", {"a1": "abc"}), + ("", {}), + (None, {}), + # 值里可以有等号,不能被截断 + ("a1=abc=def", {"a1": "abc=def"}), + ], +) +def test_trans_cookies(raw, expected): + assert trans_cookies(raw) == expected + + +def test_client_without_a1_cannot_sign(): + """a1 参与签名,没有它连请求都发不出去 —— 要提前拦而不是发出去再猜。""" + assert CreatorClient("web_session=abc").looks_authenticated is False + assert CreatorClient("a1=abc").looks_authenticated is True + + +# --- 数值解析 -------------------------------------------------------------- +# +# 后台返回的可能是数字,也可能是 "1.2万" / "12.3%" / "1分30秒" 这类展示值。 +# 解析不出来一律 None —— 不是 0。0 是真实值,None 是"不知道"。 + + +@pytest.mark.parametrize( + "raw, expected", + [ + (123, 123), + ("123", 123), + ("1,234", 1234), + ("1.2万", 12000), + ("3万", 30000), + ("1.5w", 15000), + ("1亿", 100000000), + (0, 0), + ("0", 0), + # 这些必须是 None 而不是 0 —— 把"没给"当成"是零"会让报表说谎 + (None, None), + ("", None), + ("-", None), + ("暂无", None), + ("abc", None), + (True, None), + ], +) +def test_as_int(raw, expected): + assert as_int(raw) == expected + + +@pytest.mark.parametrize( + "raw, expected", + [ + (12.3, 12.3), + ("12.3%", 12.3), + ("12.3", 12.3), + (None, None), + ("-", None), + ("暂无数据", None), + ], +) +def test_as_float(raw, expected): + assert as_float(raw) == expected + + +@pytest.mark.parametrize( + "raw, expected", + [ + (45, 45.0), + ("45", 45.0), + ("1分30秒", 90.0), + ("2分", 120.0), + ("30秒", 30.0), + ("01:30", 90.0), + ("1:00:00", 3600.0), + (None, None), + ("-", None), + ("abc", None), + ], +) +def test_as_seconds(raw, expected): + assert as_seconds(raw) == expected + + +# --- 字段归一化 ------------------------------------------------------------ + + +def test_normalize_note_maps_aliases(): + """不同来源的记录用不同字段名,别名表要能都接住。""" + note = normalize_note( + { + "note_id": "abc123", + "title": "标题", + "publish_time": 1700000000000, + "view_count": "1.2万", + "like_count": 34, + "collected_count": 5, + "share_count": 2, + "comment_count": 7, + "cover_click_rate": "12.5%", + "avg_watch_time": "1分30秒", + } + ) + + assert note["note_id"] == "abc123" + assert note["title"] == "标题" + assert note["views"] == 12000 + assert note["likes"] == 34 + assert note["favorites"] == 5 + assert note["shares"] == 2 + assert note["comments"] == 7 + assert note["cover_ctr"] == 12.5 + assert note["avg_watch_seconds"] == 90.0 + + +def test_normalize_note_leaves_missing_fields_as_none(): + note = normalize_note({"note_id": "abc123"}) + + assert note["note_id"] == "abc123" + assert note["views"] is None + assert note["likes"] is None + + +def test_find_note_list_digs_the_array_out_of_a_nested_payload(): + """接口的确切结构没见过,所以按"像是一批笔记记录"来找,不写死路径。""" + payload = { + "code": 0, + "data": { + "result": { + "success": True, + "notes": [ + {"note_id": "n1", "views": 10, "likes": 1}, + {"note_id": "n2", "views": 20, "likes": 2}, + ], + } + }, + } + + found = find_note_list(payload) + + assert [item["note_id"] for item in found] == ["n1", "n2"] + + +def test_find_note_list_returns_empty_for_the_permission_gated_envelope(): + """权限未生效时接口返回的就是这个 —— 必须安静地给出空列表,不是报错。""" + payload = { + "code": 0, + "success": True, + "msg": "成功", + "data": {"result": {"success": True, "code": 0, "message": "success"}}, + } + + assert find_note_list(payload) == [] + + +# --- 签名 ------------------------------------------------------------------ + + +def test_signed_api_carries_the_url_prefix(): + """待签字符串必须带 `url=`。少了它网关返回 406,而 406 的响应体看不出错在哪。""" + assert signing.signed_api("/api/galaxy/user/info") == "url=/api/galaxy/user/info" + assert ( + signing.signed_api("/api/x", "a=1&b=2") + == "url=/api/x?a=1&b=2" + ) + + +def test_sign_returns_xs_and_xt(): + headers = signing.sign_xyw("url=/api/galaxy/user/info", "some-a1") + + assert set(headers) == {"x-s", "x-t"} + assert headers["x-s"].startswith("XYW_") + assert headers["x-t"].isdigit() + + +def test_signature_is_stable_for_a_fixed_timestamp(): + """同一输入同一时间戳必须得到同一签名 —— 否则说明有隐藏的随机源。""" + first = signing.sign_xyw("url=/api/x", "a1", timestamp_ms=1700000000000) + second = signing.sign_xyw("url=/api/x", "a1", timestamp_ms=1700000000000) + + assert first == second + + +def test_signature_changes_with_the_signed_string(): + """签名必须真的绑定待签内容,否则改参数不会被发现 —— 那这个签名就没意义了。""" + base = signing.sign_xyw("url=/api/x?a=1", "a1", timestamp_ms=1700000000000) + other = signing.sign_xyw("url=/api/x?a=2", "a1", timestamp_ms=1700000000000) + + assert base["x-s"] != other["x-s"] + + +# --- 凭证不外泄 ------------------------------------------------------------ + + +def test_account_dict_never_carries_the_cookie(): + """cookie 等于登录态。对外结构里只该有 `has_cookie`。""" + account = CreatorAccount( + id=1, + nickname="测试", + user_id="u1", + cookie="a1=SECRET; web_session=SECRET", + created_at=0, + updated_at=0, + ) + + payload = _account_dict(account) + + assert payload["has_cookie"] is True + assert "cookie" not in payload + assert "SECRET" not in str(payload) diff --git a/webui/src/App.tsx b/webui/src/App.tsx index 871195f..b4cb67a 100644 --- a/webui/src/App.tsx +++ b/webui/src/App.tsx @@ -5,13 +5,14 @@ import { Sidebar } from '@/components/layout/Sidebar' import { MainContent } from '@/components/layout/MainContent' import { CrawlerConfigPanel } from '@/components/config/CrawlerConfigPanel' import { MonitorDashboard } from '@/components/monitor/MonitorDashboard' +import { OperationView } from '@/components/creator/OperationView' import { ReportView } from '@/components/monitor/ReportView' import { SettingsView } from '@/components/settings/SettingsView' import { Login } from '@/components/auth/Login' import { EnvironmentCheck, isEnvChecked } from '@/components/env/EnvironmentCheck' import { authApi, setUnauthorizedHandler } from '@/lib/api' -export type AppView = 'crawler' | 'monitor' | 'report' | 'settings' +export type AppView = 'crawler' | 'monitor' | 'operation' | 'report' | 'settings' function App() { // null = still probing. Rendering the app while unknown would briefly mount @@ -91,6 +92,7 @@ function App() { )} {view === 'monitor' && } + {view === 'operation' && } {view === 'report' && } {view === 'settings' && } diff --git a/webui/src/components/creator/AddAccountDialog.tsx b/webui/src/components/creator/AddAccountDialog.tsx new file mode 100644 index 0000000..408f6e3 --- /dev/null +++ b/webui/src/components/creator/AddAccountDialog.tsx @@ -0,0 +1,150 @@ +import { useEffect, useState } from 'react' +import { useQueryClient } from '@tanstack/react-query' +import { AlertTriangle, CheckCircle2, Loader2, QrCode, RefreshCw, X } from 'lucide-react' + +import { Button } from '@/components/ui/button' +import { + Dialog, + DialogContent, + DialogDescription, + DialogHeader, + DialogTitle, +} from '@/components/ui/dialog' +import { + useCancelCreatorLogin, + useCreatorLoginStatus, + useStartCreatorLogin, +} from '@/hooks/useCreator' + +/** + * 扫码新增一个运营账号。 + * + * 后端给每个账号开一个**临时浏览器上下文**,扫完取出 cookie 就丢弃,所以: + * 登第二个账号不会把第一个顶掉,也不会影响监控那个登录态。 + */ +export function AddAccountDialog({ + open, + onOpenChange, +}: { + open: boolean + onOpenChange: (open: boolean) => void +}) { + const queryClient = useQueryClient() + const [polling, setPolling] = useState(false) + + const { data: state } = useCreatorLoginStatus(polling) + const start = useStartCreatorLogin() + const cancel = useCancelCreatorLogin() + + const status = state?.status ?? 'idle' + + useEffect(() => { + if (!open) return + if (status === 'waiting' || status === 'idle') return + setPolling(false) + if (status === 'success') { + queryClient.invalidateQueries({ queryKey: ['creatorAccounts'] }) + } + }, [status, open, queryClient]) + + // 关掉弹窗时把后端那个临时上下文也收掉,别让它挂在浏览器里。 + const handleOpenChange = (next: boolean) => { + if (!next) { + setPolling(false) + if (status === 'waiting') cancel.mutate() + } + onOpenChange(next) + } + + const begin = () => { + setPolling(true) + start.mutate() + } + + const busy = start.isPending || cancel.isPending + + return ( + + + + 新增运营账号 + + 用小红书 App 扫码登录要添加的账号。每个账号使用独立的临时会话,互不影响。 + + + +
+ {status === 'waiting' && state?.image && ( +
+ {/* 二维码直接来自页面,是个 data: URL —— 不经过第三方,也不在服务器上落盘 */} + 登录二维码 +

+ 剩余 {state.expires_in} 秒 +

+ +
+ )} + + {status === 'waiting' && !state?.image && ( +

+ + 正在从浏览器取二维码… +

+ )} + + {status === 'success' && ( +

+ + {state?.message || '账号已添加'} +

+ )} + + {(status === 'error' || status === 'expired') && ( +
+

+ + {state?.message || '获取二维码失败'} +

+ +
+ )} + + {status === 'idle' && ( +
+

+ 需要先在「系统设置」里打开 接管已有 Chrome(CDP), + 并确保那台 Chrome 正以 9222 端口运行。 +

+ +
+ )} +
+
+
+ ) +} diff --git a/webui/src/components/creator/OperationView.tsx b/webui/src/components/creator/OperationView.tsx new file mode 100644 index 0000000..d64b3d6 --- /dev/null +++ b/webui/src/components/creator/OperationView.tsx @@ -0,0 +1,343 @@ +import { useState } from 'react' +import { + AlertTriangle, + ArrowLeft, + Clock, + Plus, + RefreshCw, + Trash2, + UserRound, +} from 'lucide-react' + +import { Badge } from '@/components/ui/badge' +import { Button } from '@/components/ui/button' +import { + useCheckCreatorAccount, + useCreatorAccount, + useCreatorAccounts, + useDeleteCreatorAccount, + useSyncCreatorAccount, +} from '@/hooks/useCreator' +import { formatCount, formatDateTime, formatRelative } from '@/lib/monitorFormat' +import type { CreatorNote, CreatorPermissionStatus } from '@/types/creator' + +import { AddAccountDialog } from './AddAccountDialog' + +const PERMISSION_LABEL: Record = { + active: { text: '数据已开通', tone: 'text-cyber-neon-green' }, + // 实测中最常见的一种,必须和"没有权限"分开说 —— 否则用户以为采集坏了, + // 其实只是在等次日生效。 + pending: { text: '权限待生效', tone: 'text-cyber-neon-orange' }, + missing: { text: '无数据权限', tone: 'text-cyber-neon-pink' }, + unknown: { text: '权限未知', tone: 'text-cyber-text-muted' }, +} + +const SUMMARY_TILES: Array<{ key: keyof CreatorNote; label: string }> = [ + { key: 'exposure', label: '曝光' }, + { key: 'views', label: '观看' }, + { key: 'likes', label: '点赞' }, + { key: 'comments', label: '评论' }, + { key: 'favorites', label: '收藏' }, + { key: 'shares', label: '分享' }, + { key: 'new_followers', label: '涨粉' }, +] + +/** 比率字段用百分号展示;其余用紧凑计数。 */ +function formatRate(value: number | null): string { + return value === null ? '—' : `${value.toFixed(1)}%` +} + +function formatSeconds(value: number | null): string { + if (value === null) return '—' + if (value < 60) return `${Math.round(value)} 秒` + return `${Math.floor(value / 60)} 分 ${Math.round(value % 60)} 秒` +} + +function AccountList({ + selectedId, + onSelect, + onAdd, +}: { + selectedId: number | null + onSelect: (id: number) => void + onAdd: () => void +}) { + // 有同步在后端跑时,列表要自己刷新,否则状态永远是旧的。 + const { data: accounts, isLoading } = useCreatorAccounts(true) + + if (isLoading) { + return

加载中…

+ } + + if (!accounts || accounts.length === 0) { + return ( +
+ +

+ 还没有运营账号,扫码添加一个 +

+ +
+ ) + } + + return ( +
+ {accounts.map((account) => { + const permission = PERMISSION_LABEL[account.permission_status] + return ( + + ) + })} +
+ ) +} + +function AccountDetail({ accountId, onBack }: { accountId: number; onBack: () => void }) { + const { data, isLoading } = useCreatorAccount(accountId) + const sync = useSyncCreatorAccount() + const check = useCheckCreatorAccount() + const remove = useDeleteCreatorAccount() + + if (isLoading || !data) { + return

加载中…

+ } + + const { account, notes, summary } = data + const permission = PERMISSION_LABEL[account.permission_status] + + return ( +
+
+ + {account.avatar && ( + + )} + + {account.nickname || '未命名账号'} + + + 小红书号 {account.red_id || '—'} + + +
+ + + +
+
+ + {/* 权限状态必须显眼:它解释了"为什么没有数据"。后台的原话照抄,不改写。 */} + {account.permission_status !== 'active' && ( +
+ +
+ {permission.text} + {account.permission_tip && ( + {account.permission_tip} + )} +
+ 数据权限由创作者后台按天开通。在此之前同步会成功但返回 0 条 —— 那是正常的, + 不是采集失败。 +
+
+
+ )} + + {account.last_error && ( +

+ + {account.last_error} +

+ )} + +
+ {SUMMARY_TILES.map((tile) => ( +
+
{tile.label}
+
+ {formatCount(summary[tile.key as keyof typeof summary] ?? 0)} +
+
+ ))} +
+ + {notes.length === 0 ? ( +

+ {account.permission_status === 'active' + ? '这个账号在所选时间范围内没有作品,或还没同步过 —— 点「同步数据」试试' + : '数据权限生效后即可看到作品数据'} +

+ ) : ( +
+ + + + + {SUMMARY_TILES.map((tile) => ( + + ))} + + + + + + + {notes.map((note) => ( + + + {SUMMARY_TILES.map((tile) => ( + + ))} + + + + + ))} + +
作品 + {tile.label} + 点击率均看时长完播率
+
+ {note.title || note.note_id} +
+
+ {note.publish_time ? formatDateTime(note.publish_time) : '—'} +
+
+ {formatCount(note[tile.key] as number | null)} + + {formatRate(note.cover_ctr)} + + {formatSeconds(note.avg_watch_seconds)} + + {formatRate(note.completion_rate)} +
+
+ )} +
+ ) +} + +/** + * 运营:管理自己的小红书账号。 + * + * 与「监控」并列而非其子视图。监控抓的是公开数据(点赞/收藏/评论/分享),这里是 + * 创作者后台的运营指标(曝光/观看/完播率/涨粉)。两者数据形状不同、凭据不同、 + * 采集方式也不同(那边要浏览器登录态,这边是纯请求)。 + */ +export function OperationView() { + const [selectedId, setSelectedId] = useState(null) + const [addOpen, setAddOpen] = useState(false) + + return ( +
+
+
+
+ 运营账号 + + {selectedId === null + ? '管理自己的小红书账号,读取创作者后台的曝光/观看/完播等运营数据' + : '点「返回」回到账号列表'} + +
+ {selectedId === null && ( + + )} +
+ + {selectedId === null ? ( + setAddOpen(true)} + /> + ) : ( + setSelectedId(null)} /> + )} +
+ + +
+ ) +} diff --git a/webui/src/components/layout/Sidebar.tsx b/webui/src/components/layout/Sidebar.tsx index 490e92b..cf10af5 100644 --- a/webui/src/components/layout/Sidebar.tsx +++ b/webui/src/components/layout/Sidebar.tsx @@ -1,5 +1,5 @@ import { useState } from 'react' -import { Bug, Wifi, BarChart3, Cog, LogOut, Radar, Settings, Terminal } from 'lucide-react' +import { Briefcase, Bug, Wifi, BarChart3, Cog, LogOut, Radar, Settings, Terminal } from 'lucide-react' import { useTranslation } from 'react-i18next' import { Badge } from '@/components/ui/badge' import { SystemSettingsDialog } from '@/components/settings/SystemSettingsDialog' @@ -19,6 +19,7 @@ interface SidebarProps { const NAV_ITEMS: Array<{ value: AppView; label: string; icon: typeof Terminal }> = [ { value: 'crawler', label: '采集', icon: Terminal }, { value: 'monitor', label: '监控', icon: Radar }, + { value: 'operation', label: '运营', icon: Briefcase }, { value: 'report', label: '报表', icon: BarChart3 }, { value: 'settings', label: '设置', icon: Settings }, ] diff --git a/webui/src/hooks/useCreator.ts b/webui/src/hooks/useCreator.ts new file mode 100644 index 0000000..cc649c2 --- /dev/null +++ b/webui/src/hooks/useCreator.ts @@ -0,0 +1,99 @@ +import { useMutation, useQuery, useQueryClient } from '@tanstack/react-query' +import { toast } from 'sonner' + +import { creatorApi } from '@/lib/api' + +const ACCOUNTS_KEY = ['creatorAccounts'] + +/** 账号列表。同步在后台跑,所以列表本身也顺带轮询,好让状态自己刷新出来。 */ +export function useCreatorAccounts(polling = false) { + return useQuery({ + queryKey: ACCOUNTS_KEY, + queryFn: async () => (await creatorApi.listAccounts()).data.accounts, + refetchInterval: polling ? 5000 : false, + }) +} + +export function useCreatorAccount(id: number | null) { + return useQuery({ + queryKey: ['creatorAccount', id], + queryFn: async () => (await creatorApi.getAccount(id as number)).data, + enabled: id !== null, + }) +} + +export function useDeleteCreatorAccount() { + const queryClient = useQueryClient() + return useMutation({ + mutationFn: (id: number) => creatorApi.deleteAccount(id), + onSuccess: () => { + toast.success('账号已删除') + queryClient.invalidateQueries({ queryKey: ACCOUNTS_KEY }) + }, + onError: (error: Error) => toast.error(`删除失败:${error.message}`), + }) +} + +export function useCheckCreatorAccount() { + const queryClient = useQueryClient() + return useMutation({ + mutationFn: (id: number) => creatorApi.checkAccount(id), + onSuccess: () => { + queryClient.invalidateQueries({ queryKey: ACCOUNTS_KEY }) + }, + onError: (error: Error) => toast.error(`检测失败:${error.message}`), + }) +} + +export function useSyncCreatorAccount() { + const queryClient = useQueryClient() + return useMutation({ + mutationFn: ({ id, days }: { id: number; days?: number }) => + creatorApi.syncAccount(id, days), + onSuccess: () => { + // 后端是后台任务,立刻重新拉一次只会看到旧状态;给用户一句"已开始", + // 列表的轮询会把结果带回来。 + toast.success('同步已开始,稍后自动刷新') + queryClient.invalidateQueries({ queryKey: ACCOUNTS_KEY }) + }, + onError: (error: Error) => toast.error(`同步失败:${error.message}`), + }) +} + +// --- 扫码新增账号 --------------------------------------------------------- + +/** 只在扫码中轮询;空闲时没必要一直问。 */ +export function useCreatorLoginStatus(polling: boolean) { + return useQuery({ + queryKey: ['creatorLogin'], + queryFn: async () => (await creatorApi.getLogin()).data, + refetchInterval: polling ? 2000 : false, + }) +} + +export function useStartCreatorLogin() { + const queryClient = useQueryClient() + return useMutation({ + mutationFn: () => creatorApi.startLogin(), + onSuccess: (response) => { + queryClient.setQueryData(['creatorLogin'], response.data) + }, + onError: (error: Error) => { + // 后端给的原因是这里唯一有价值的信息(连不上 Chrome、页面上没有二维码), + // 而 axios 会把它压成 "Request failed with status code 502"。 + const detail = (error as { response?: { data?: { detail?: string } } })?.response?.data + ?.detail + toast.error(`获取二维码失败:${detail ?? error.message}`) + }, + }) +} + +export function useCancelCreatorLogin() { + const queryClient = useQueryClient() + return useMutation({ + mutationFn: () => creatorApi.cancelLogin(), + onSuccess: (response) => { + queryClient.setQueryData(['creatorLogin'], response.data) + }, + }) +} diff --git a/webui/src/lib/api.ts b/webui/src/lib/api.ts index f74f8e7..eaf1059 100644 --- a/webui/src/lib/api.ts +++ b/webui/src/lib/api.ts @@ -18,6 +18,12 @@ import type { TaskCreatePayload, WebhookStatus, } from '@/types/monitor' +import type { + CreatorAccount, + CreatorAccountDetail, + CreatorLoginState, + CreatorSyncResult, +} from '@/types/creator' const api = axios.create({ baseURL: '/api', @@ -274,4 +280,23 @@ export const monitorApi = { testWebhook: (url?: string) => api.post('/monitor/webhook/test', { url: url ?? null }), } +/** + * 运营模块。与 `monitorApi` 并列 —— 它管的是自己的账号,走的是纯请求的创作者后台, + * 和公开数据监控不是一回事。 + */ +export const creatorApi = { + listAccounts: () => api.get<{ accounts: CreatorAccount[] }>('/creator/accounts'), + getAccount: (id: number) => api.get(`/creator/accounts/${id}`), + deleteAccount: (id: number) => api.delete(`/creator/accounts/${id}`), + checkAccount: (id: number) => api.post(`/creator/accounts/${id}/check`), + // 同步在后端后台跑(分页 + 节流可能几分钟),这里只负责触发。 + syncAccount: (id: number, days = 90) => + api.post(`/creator/accounts/${id}/sync`, null, { params: { days } }), + + // 扫码新增账号。cookie 只在后端内存里流转,不会出现在这些响应里。 + startLogin: () => api.post('/creator/login'), + getLogin: () => api.get('/creator/login'), + cancelLogin: () => api.delete('/creator/login'), +} + export default api diff --git a/webui/src/types/creator.ts b/webui/src/types/creator.ts new file mode 100644 index 0000000..38899b9 --- /dev/null +++ b/webui/src/types/creator.ts @@ -0,0 +1,89 @@ +/** 运营模块:自己的小红书账号,以及创作者后台给的数据。 */ + +/** + * 数据权限状态。 + * + * `pending` 是实测中最常见的一种:后台原话是「已为您申请数据权限,次日可查看」。 + * 它必须和「没有权限」分开显示 —— 否则用户会以为采集坏了,其实只是在等审批。 + */ +export type CreatorPermissionStatus = 'unknown' | 'pending' | 'active' | 'missing' + +export type CreatorAccountStatus = 'ok' | 'expired' | 'error' + +/** 注意:接口**不下发 cookie**,只给 `has_cookie`。 */ +export interface CreatorAccount { + id: number + nickname: string + user_id: string + red_id: string + avatar: string + status: CreatorAccountStatus + permission_status: CreatorPermissionStatus + /** 后台原话,照抄不改写。 */ + permission_tip: string + last_checked_at: number | null + last_synced_at: number | null + last_error: string | null + has_cookie: boolean + created_at: number + note_count: number +} + +/** + * 一篇作品的运营数据。 + * + * 除时间外全部可为 null:接口没给、或解析不出来,都存 null 而不是 0 —— + * 0 是真实值,null 是"不知道",混在一起报表会说谎。 + */ +export interface CreatorNote { + note_id: string + title: string + publish_time: number | null + exposure: number | null + views: number | null + likes: number | null + comments: number | null + favorites: number | null + shares: number | null + new_followers: number | null + danmaku: number | null + cover_ctr: number | null + avg_watch_seconds: number | null + two_second_exit_rate: number | null + completion_rate: number | null + captured_at: number +} + +export interface CreatorSummary { + exposure: number + views: number + likes: number + comments: number + favorites: number + shares: number + new_followers: number +} + +export interface CreatorAccountDetail { + account: CreatorAccount + notes: CreatorNote[] + summary: CreatorSummary +} + +export type CreatorLoginStatus = 'idle' | 'waiting' | 'success' | 'expired' | 'error' + +export interface CreatorLoginState { + status: CreatorLoginStatus + message: string + /** `data:image/...;base64,...`, empty unless status is `waiting`. */ + image: string + elapsed: number + expires_in: number + /** 扫码成功后后端已落库的账号。 */ + account: CreatorAccount | null +} + +export interface CreatorSyncResult { + message: string + days: number +}