Files
MediaCrawler/tools/probe_creator_api.py
butubb 2613f7577f
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
feat(creator): Phase 0 探针 —— 创作者后台数据可以纯请求拿到
结论:签名可自造、主站 cookie 即可认证、接口与参数已与真实页面对齐。不需要浏览器、不需要独立的创作者登录。

- tools/probe_creator_api.py: 纯 HTTP 探针。用 XYW_ 算法自签(MD5 → base64 → AES-128-CBC,
  密钥与 IV 与 xhshow/config/config.py 逐字节一致),对 note/analyze/list 发请求
- tools/probe_creator_page.py: 打开真实数据分析页,记录页面自己发的请求,作为地面真相

Phase 0 的三条实测结论:
1. 签名可伪造。三种写法里只有「url= + 路径 + 查询串」被接受(200);仅路径、或裸路径都 406。
   并且不带 cookie 时返回的是应用层 401「无登录信息」而非网关 406 —— 说明签名每次都已通过
2. 主站 .xiaohongshu.com 的 cookie 就能认证创作者后台,不需要单独的创作者会话
3. 当前账号 dfg 返回空数据不是技术问题:permission/query 的 tip_msg 是
   「已为您申请数据权限,次日可查看」,display/status 均为 0,即权限尚未生效

关键佐证:真实页面调 note/analyze/list 用的查询串与本探针生成的完全一致,且拿到同一份空响应。
2026-10-07 16:22:15 +08:00

211 lines
7.9 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Phase 0 probe: can the creator backend be reached with a plain signed request?
The question this answers, and why it is worth a whole script: two sources
disagree. The `xhshow` library ships `sign_xyw()` whose docstring says it exists
because the creator data APIs *reject* the main-site signature with HTTP 406 -- but
a field report claims the creator gateway rejects *any* self-made request with 406
regardless of signature. Only a live request settles it.
The response is read as a three-way verdict, because a bare "it failed" is not
useful here:
406 -> the gateway rejected the signature. The browser-interception route is
the only way forward.
401 / not-logged-in
-> the signature PASSED and only the creator session is missing. That is
good news: it means Phase 1 is pure-request after all.
200 -> we are through, and the payload is captured for field mapping.
Cookies are read out of the browser over CDP and never printed -- they are
credentials, and this script has no reason to echo them.
"""
import argparse
import base64
import hashlib
import json
import sys
import urllib.parse
from datetime import datetime, time, timedelta
# The XYW_ scheme, matching both xhshow/config/config.py and the independent
# reverse-engineering in xiaohongshu-cli. Constants are byte-identical in both.
XYW_AES_KEY = b"7cc4adla5ay0701v"
XYW_AES_IV = b"4uzjr7mbsibcaldp"
XYW_ENV_FLAGS = "0|0|0|1|0|0|1|0|0|0|1|0|0|0|0|1|0|0|0"
CREATOR_ORIGIN = "https://creator.xiaohongshu.com"
# The data-analysis note list. This is the page the creator console itself calls,
# and it is where exposure/views live.
NOTE_LIST_PATH = "/api/galaxy/creator/datacenter/note/analyze/list"
USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36"
)
def _aes_encrypt_hex(plaintext: str) -> str:
from Crypto.Cipher import AES
from Crypto.Util.Padding import pad
cipher = AES.new(XYW_AES_KEY, AES.MODE_CBC, XYW_AES_IV)
return cipher.encrypt(pad(plaintext.encode("utf-8"), AES.block_size)).hex()
def sign_xyw(api: str, a1: str, app_id: str = "ugc", data: dict | None = None) -> dict[str, str]:
"""Mirror of xiaohongshu-cli's creator_signing.sign_creator.
Written out rather than imported so the probe can vary ``api`` and ``app_id``
independently -- figuring out which combination the gateway accepts is the
entire point of running it.
"""
content = api
if data is not None:
content += json.dumps(data, separators=(",", ":"), ensure_ascii=False)
digest = hashlib.md5(content.encode("utf-8")).hexdigest()
timestamp_ms = int(datetime.now().timestamp() * 1000)
plaintext = f"x1={digest};x2={XYW_ENV_FLAGS};x3={a1};x4={timestamp_ms};"
encoded = base64.b64encode(plaintext.encode("utf-8")).decode("utf-8")
envelope = {
"signSvn": "56",
"signType": "x2",
"appId": app_id,
"signVersion": "1",
"payload": _aes_encrypt_hex(encoded),
}
x_s = "XYW_" + base64.b64encode(
json.dumps(envelope, separators=(",", ":")).encode("utf-8")
).decode("utf-8")
return {"x-s": x_s, "x-t": str(timestamp_ms)}
def build_query(start_days_ago: int, end_days_ago: int, page_size: int = 10) -> str:
"""Query string exactly as the working collector builds it: epoch millis."""
today = datetime.now().replace(hour=0, minute=0, second=0, microsecond=0)
def ms(days_ago: int, at_end: bool) -> int:
day = (today - timedelta(days=days_ago)).date()
clock = time(23, 59, 59) if at_end else time(0, 0, 0)
return int(datetime.combine(day, clock).timestamp() * 1000)
return urllib.parse.urlencode(
{
"post_begin_time": ms(start_days_ago, False),
"post_end_time": ms(end_days_ago, True),
"type": 0,
"page_size": page_size,
"page_num": 1,
}
)
async def cookies_from_cdp() -> dict[str, str]:
"""Session cookies for creator.xiaohongshu.com, read out of the live browser."""
from playwright.async_api import async_playwright
playwright = await async_playwright().start()
try:
browser = await playwright.chromium.connect_over_cdp(
"http://127.0.0.1:9222", timeout=15000
)
jars = {}
for context in browser.contexts:
for cookie in await context.cookies():
jars[cookie["name"]] = cookie["value"]
return jars
finally:
await playwright.stop()
def classify(status: int, body: str) -> str:
if status == 406:
return "406 —— 网关拒了签名(回退浏览器拦截路线)"
if status == 200:
return "200 —— 通了,可以考虑解析数据"
if status in (401, 403):
return f"{status} —— 签名过了,只差创作者会话(好消息)"
return f"{status} —— 未知,需要看响应体"
async def main() -> int:
import httpx
ap = argparse.ArgumentParser()
ap.add_argument("--start-days-ago", type=int, default=30)
ap.add_argument("--end-days-ago", type=int, default=0)
ap.add_argument("--app-id", default="ugc", help="参考实现用 ugc;xhshow 默认 xhs-pc-web")
ap.add_argument("--cookie", default="", help="留空则从 CDP 浏览器读取")
ap.add_argument("--show-body", action="store_true", help="打印响应前 800 字符")
ap.add_argument(
"--no-cookie-header",
action="store_true",
help="签名照签(仍需 a1)但不发 cookie 头,用来分清"
"「空数据是缺会话」还是「接口本身就这样」",
)
args = ap.parse_args()
if args.cookie:
cookies = dict(
pair.split("=", 1) for pair in args.cookie.split("; ") if "=" in pair
)
else:
cookies = await cookies_from_cdp()
print(f" cookie 条数 {len(cookies)},名字: {sorted(cookies)}")
if not cookies.get("a1"):
print(" ✗ 没有 a1 —— 签名必须用它,无法继续")
return 2
query = build_query(args.start_days_ago, args.end_days_ago)
cookie_header = "; ".join(f"{k}={v}" for k, v in cookies.items())
headers_common = {
"user-agent": USER_AGENT,
"accept": "application/json, text/plain, */*",
"origin": CREATOR_ORIGIN,
"referer": f"{CREATOR_ORIGIN}/statistics/data-analysis",
"accept-language": "zh-CN,zh;q=0.9",
}
if not args.no_cookie_header:
headers_common["cookie"] = cookie_header
# Three signing variants: the reference bakes the query into the signed string,
# but the exact form is not documented beyond an example with no query at all.
variants = {
"path+q(参考实现写法)": f"url={NOTE_LIST_PATH}?{query}",
"path only": f"url={NOTE_LIST_PATH}",
"裸 path+query(无 url= 前缀)": f"{NOTE_LIST_PATH}?{query}",
}
url = f"{CREATOR_ORIGIN}{NOTE_LIST_PATH}?{query}"
async with httpx.AsyncClient(timeout=25, follow_redirects=False) as client:
for label, api in variants.items():
signature = sign_xyw(api, cookies["a1"], app_id=args.app_id)
headers = {**headers_common, **signature}
try:
response = await client.get(url, headers=headers)
except Exception as exc: # noqa: BLE001
print(f" [{label}] 请求异常: {exc.__class__.__name__}: {exc}")
continue
print(f"\n [{label}]")
print(f" HTTP {response.status_code} {classify(response.status_code, response.text)}")
body = response.text or ""
if body:
print(f" 响应前 160 字符: {body[:160]!r}")
if args.show_body and body:
print(f" 完整响应: {body[:800]}")
print("\n 提示:若三种都返回 406,再试 --app-id xhs-pc-web。")
return 0
if __name__ == "__main__":
import asyncio
sys.exit(asyncio.run(main()))