# -*- coding: utf-8 -*- # Copyright (c) 2025 relakkes@gmail.com # # This file is part of MediaCrawler project. # Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/creator/client.py # GitHub: https://github.com/NanmiCoder # Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 # # 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: # 1. 不得用于任何商业用途。 # 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 # 3. 不得进行大规模爬取或对平台造成运营干扰。 # 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 # 5. 不得用于任何非法或不当的用途。 # # 详细许可条款请参阅项目根目录下的LICENSE文件。 # 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 """小红书创作者后台的纯请求客户端。 无浏览器:签名在本地算(见 signing.py),请求走 httpx。 **关于字段名的谨慎**:Phase 0 抓到的响应里,列表接口因为账号权限未生效而返回空壳 (`data.result` 只有 `{success, code, message}` 没有数据),所以**真实字段名尚未亲眼 见过**。因此每个指标都写成**多别名匹配**,并且解析不出来时存 `None` 而不是 0 —— 0 是真实值,None 是"不知道",两者混淆会让报表说谎。 """ import re from typing import Any, Dict, List, Optional import httpx from .signing import sign_xyw, signed_api CREATOR_ORIGIN = "https://creator.xiaohongshu.com" DATA_ANALYSIS_PAGE = f"{CREATOR_ORIGIN}/statistics/data-analysis" USER_INFO_PATH = "/api/galaxy/user/info" PERMISSION_PATH = "/api/galaxy/creator/datacenter/permission/query" NOTE_LIST_PATH = "/api/galaxy/creator/datacenter/note/analyze/list" USER_AGENT = ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36" ) # 签名被网关拒绝时返回的响应体。区分它和普通业务错误很重要:406 说明签名写错了, # 而应用层的 code/-100 说明签名没问题、只是没有登录态。 SIGNATURE_REJECTED_MARKERS = ("code", -1) class CreatorApiError(RuntimeError): """调用创作者后台失败。``status`` 用于区分是网关拒绝还是业务错误。""" def __init__(self, message: str, status: int = 0, payload: Any = None): super().__init__(message) self.status = status self.payload = payload def trans_cookies(cookie_str: str) -> Dict[str, str]: """把 cookie 字符串解析成字典。容忍末尾分号、换行和零散空格。""" jar: Dict[str, str] = {} for chunk in re.split(r"[;\n]", cookie_str or ""): chunk = chunk.strip() if not chunk or "=" not in chunk: continue name, value = chunk.split("=", 1) name = name.strip() if name: jar[name] = value.strip() return jar def _pick(item: Dict[str, Any], *names: str) -> Any: """按别名顺序取第一个存在的键。 字段名来自二手资料,未亲眼验证,所以不赌单一命名。 """ for name in names: if name in item and item[name] is not None: return item[name] return None _COUNT_UNITS = {"万": 10_000, "w": 10_000, "W": 10_000, "亿": 100_000_000, "k": 1_000, "K": 1_000} def as_int(value: Any) -> Optional[int]: """解析计数。处理 "1.2万"、"1,234"、"123" 与已经是数字的情况。""" if value is None or isinstance(value, bool): return None if isinstance(value, (int, float)): return int(value) text = str(value).strip().replace(",", "").replace(" ", "") if not text or text in ("-", "--", "暂无"): return None unit = 1 suffix = text[-1] if suffix in _COUNT_UNITS: unit = _COUNT_UNITS[suffix] text = text[:-1] try: return int(float(text) * unit) except ValueError: return None def as_float(value: Any) -> Optional[float]: """解析比率。``"12.3%"`` -> 12.3;``"0.123"`` 原样返回数字。""" if value is None or isinstance(value, bool): return None if isinstance(value, (int, float)): return float(value) text = str(value).strip().replace("%", "") try: return float(text) except ValueError: return None _DURATION_RE = re.compile(r"(?:(\d+)\s*分)?\s*(?:(\d+)\s*秒)?") def as_seconds(value: Any) -> Optional[float]: """解析时长。处理 ``"1分30秒"``、``"01:30"``、``"45"``(秒)。""" if value is None or isinstance(value, bool): return None if isinstance(value, (int, float)): return float(value) text = str(value).strip() if not text or text in ("-", "--"): return None if ":" in text: parts = text.split(":") try: total = 0.0 for part in parts: total = total * 60 + float(part) return total except ValueError: return None if "分" in text or "秒" in text: minutes = re.search(r"(\d+)\s*分", text) seconds = re.search(r"(\d+)\s*秒", text) if not minutes and not seconds: return None return float(minutes.group(1) if minutes else 0) * 60 + float( seconds.group(1) if seconds else 0 ) try: return float(text) except ValueError: return None # 指标 -> 候选原始字段名。别名来自公开资料,权威与否只能等真实响应来验证。 _FIELD_ALIASES: Dict[str, tuple] = { "exposure": ("exposure", "exposure_count", "imp", "impression", "impression_count"), "views": ("views", "view", "view_count", "watch", "watch_count", "read_count"), "likes": ("likes", "like", "like_count", "liked_count"), "comments": ("comments", "comment", "comment_count", "comments_count"), "favorites": ("favorites", "favorite", "favorite_count", "collect", "collect_count", "collected_count"), "shares": ("shares", "share", "share_count", "shared_count"), "new_followers": ("new_followers", "fans_growth", "follower_growth", "increase_fans"), "danmaku": ("danmaku", "danmaku_count", "barrage"), "cover_ctr": ("cover_ctr", "cover_click_rate", "cover_click_ratio", "ctr"), "avg_watch_seconds": ("avg_watch_seconds", "avg_watch_time", "average_watch_time"), "two_second_exit_rate": ("two_second_exit_rate", "2s_exit_rate", "exit_rate_2s"), "completion_rate": ("completion_rate", "finish_rate", "complete_rate"), } _COUNT_FIELDS = { "exposure", "views", "likes", "comments", "favorites", "shares", "new_followers", "danmaku", } _DURATION_FIELDS = {"avg_watch_seconds"} _RATE_FIELDS = {"two_second_exit_rate", "completion_rate", "cover_ctr"} NOTE_ID_ALIASES = ("note_id", "noteId", "id", "content_id", "item_id") TITLE_ALIASES = ("title", "content", "note_title", "display_title") PUBLISH_TIME_ALIASES = ("publish_time", "publishTime", "post_time", "create_time", "time") def _normalize_metric(name: str, raw: Any) -> Any: if name in _COUNT_FIELDS: return as_int(raw) if name in _DURATION_FIELDS: return as_seconds(raw) if name in _RATE_FIELDS: return as_float(raw) return raw def normalize_note(item: Dict[str, Any]) -> Dict[str, Any]: """把一条原始记录规范化成落库用的字段。""" note: Dict[str, Any] = { "note_id": str(_pick(item, *NOTE_ID_ALIASES) or ""), "title": str(_pick(item, *TITLE_ALIASES) or ""), "publish_time": as_int(_pick(item, *PUBLISH_TIME_ALIASES)), } for name, aliases in _FIELD_ALIASES.items(): note[name] = _normalize_metric(name, _pick(item, *aliases)) return note def find_note_list(payload: Any) -> List[Dict[str, Any]]: """在响应里找出笔记数组。 接口的**确切结构还没亲眼见过**(权限未生效时 `data.result` 里没有数据), 所以不写死路径:遍历 JSON,挑出"看起来像一批笔记记录"的那个列表 —— 元素是 dict,且至少带一个指标字段。找不到就返回空列表,让上层如实报"没数据"。 """ best: List[Dict[str, Any]] = [] metric_keys = {alias for aliases in _FIELD_ALIASES.values() for alias in aliases} def walk(value: Any) -> None: nonlocal best if isinstance(value, dict): for child in value.values(): walk(child) elif isinstance(value, list): if value and isinstance(value[0], dict): keys = set(value[0].keys()) if keys & metric_keys and len(value) > len(best): best = value for child in value: walk(child) walk(payload) return best class CreatorClient: """一个账号的客户端。``cookie`` 就是它的全部身份。""" def __init__(self, cookie: str, timeout: float = 25.0): self.cookies = trans_cookies(cookie) self.a1 = self.cookies.get("a1", "") self._timeout = timeout @property def looks_authenticated(self) -> bool: """签名需要 a1;没有它连请求都签不出来。""" return bool(self.a1) def _headers(self, api: str, body: dict | None = None) -> Dict[str, str]: cookie_header = "; ".join(f"{k}={v}" for k, v in self.cookies.items()) return { "user-agent": USER_AGENT, "accept": "application/json, text/plain, */*", "accept-language": "zh-CN,zh;q=0.9", "origin": CREATOR_ORIGIN, "referer": DATA_ANALYSIS_PAGE, "cookie": cookie_header, **sign_xyw(api, self.a1, body=body), } async def _get(self, path: str, params: Dict[str, Any]) -> Dict[str, Any]: if not self.looks_authenticated: raise CreatorApiError("cookie 里没有 a1,无法完成签名", status=0) query = "&".join(f"{k}={v}" for k, v in params.items()) api = signed_api(path, query) url = f"{CREATOR_ORIGIN}{path}?{query}" if query else f"{CREATOR_ORIGIN}{path}" async with httpx.AsyncClient(timeout=self._timeout, follow_redirects=False) as client: response = await client.get(url, headers=self._headers(api)) return self._unwrap(response) @staticmethod def _unwrap(response: httpx.Response) -> Dict[str, Any]: if response.status_code == 406: raise CreatorApiError( "签名被网关拒绝(406)—— 待签字符串的拼法不对", status=406 ) try: payload = response.json() except Exception as exc: # noqa: BLE001 raise CreatorApiError( f"响应不是 JSON(HTTP {response.status_code})", status=response.status_code ) from exc if response.status_code == 401 or payload.get("code") == -100: raise CreatorApiError("登录态无效或已过期", status=401, payload=payload) if not payload.get("success", True): raise CreatorApiError( str(payload.get("msg") or "接口返回失败"), status=response.status_code, payload=payload, ) return payload async def fetch_user_info(self) -> Dict[str, Any]: """当前 cookie 属于哪个账号。登录后用它取名与去重。""" payload = await self._get(USER_INFO_PATH, {}) data = payload.get("data") or {} return { "user_id": str(data.get("userId") or ""), "nickname": str(data.get("userName") or ""), "avatar": str(data.get("userAvatar") or ""), "red_id": str(data.get("redId") or ""), "role": str(data.get("role") or ""), "permissions": list(data.get("permissions") or []), } async def fetch_permission(self) -> Dict[str, Any]: """数据权限状态。 ``tip_msg`` 是后台原话(实测是"已为您申请数据权限,次日可查看"),照抄不改写 —— 这条信息必须原样交给用户,它解释了"为什么没有数据"。 """ payload = await self._get(PERMISSION_PATH, {}) data = payload.get("data") or {} return { "display": data.get("display"), "status": data.get("status"), "tip": str(data.get("tip_msg") or ""), } async def fetch_note_list( self, start_ms: int, end_ms: int, page_num: int = 1, page_size: int = 10, note_type: int = 0 ) -> List[Dict[str, Any]]: """按发布时间区间取笔记列表。 参数与顺序**照抄真实页面的请求**(见 Phase 0 抓包),不要凭感觉改。 """ payload = await self._get( NOTE_LIST_PATH, { "post_begin_time": start_ms, "post_end_time": end_ms, "type": note_type, "page_size": page_size, "page_num": page_num, }, ) return [normalize_note(item) for item in find_note_list(payload)]