feat(creator): Phase 0 探针 —— 创作者后台数据可以纯请求拿到
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s

结论:签名可自造、主站 cookie 即可认证、接口与参数已与真实页面对齐。不需要浏览器、不需要独立的创作者登录。

- tools/probe_creator_api.py: 纯 HTTP 探针。用 XYW_ 算法自签(MD5 → base64 → AES-128-CBC,
  密钥与 IV 与 xhshow/config/config.py 逐字节一致),对 note/analyze/list 发请求
- tools/probe_creator_page.py: 打开真实数据分析页,记录页面自己发的请求,作为地面真相

Phase 0 的三条实测结论:
1. 签名可伪造。三种写法里只有「url= + 路径 + 查询串」被接受(200);仅路径、或裸路径都 406。
   并且不带 cookie 时返回的是应用层 401「无登录信息」而非网关 406 —— 说明签名每次都已通过
2. 主站 .xiaohongshu.com 的 cookie 就能认证创作者后台,不需要单独的创作者会话
3. 当前账号 dfg 返回空数据不是技术问题:permission/query 的 tip_msg 是
   「已为您申请数据权限,次日可查看」,display/status 均为 0,即权限尚未生效

关键佐证:真实页面调 note/analyze/list 用的查询串与本探针生成的完全一致,且拿到同一份空响应。
This commit is contained in:
2026-10-07 16:22:15 +08:00
parent 2b9ebdad87
commit 2613f7577f
2 changed files with 354 additions and 0 deletions
+210
View File
@@ -0,0 +1,210 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Phase 0 probe: can the creator backend be reached with a plain signed request?
The question this answers, and why it is worth a whole script: two sources
disagree. The `xhshow` library ships `sign_xyw()` whose docstring says it exists
because the creator data APIs *reject* the main-site signature with HTTP 406 -- but
a field report claims the creator gateway rejects *any* self-made request with 406
regardless of signature. Only a live request settles it.
The response is read as a three-way verdict, because a bare "it failed" is not
useful here:
406 -> the gateway rejected the signature. The browser-interception route is
the only way forward.
401 / not-logged-in
-> the signature PASSED and only the creator session is missing. That is
good news: it means Phase 1 is pure-request after all.
200 -> we are through, and the payload is captured for field mapping.
Cookies are read out of the browser over CDP and never printed -- they are
credentials, and this script has no reason to echo them.
"""
import argparse
import base64
import hashlib
import json
import sys
import urllib.parse
from datetime import datetime, time, timedelta
# The XYW_ scheme, matching both xhshow/config/config.py and the independent
# reverse-engineering in xiaohongshu-cli. Constants are byte-identical in both.
XYW_AES_KEY = b"7cc4adla5ay0701v"
XYW_AES_IV = b"4uzjr7mbsibcaldp"
XYW_ENV_FLAGS = "0|0|0|1|0|0|1|0|0|0|1|0|0|0|0|1|0|0|0"
CREATOR_ORIGIN = "https://creator.xiaohongshu.com"
# The data-analysis note list. This is the page the creator console itself calls,
# and it is where exposure/views live.
NOTE_LIST_PATH = "/api/galaxy/creator/datacenter/note/analyze/list"
USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36"
)
def _aes_encrypt_hex(plaintext: str) -> str:
from Crypto.Cipher import AES
from Crypto.Util.Padding import pad
cipher = AES.new(XYW_AES_KEY, AES.MODE_CBC, XYW_AES_IV)
return cipher.encrypt(pad(plaintext.encode("utf-8"), AES.block_size)).hex()
def sign_xyw(api: str, a1: str, app_id: str = "ugc", data: dict | None = None) -> dict[str, str]:
"""Mirror of xiaohongshu-cli's creator_signing.sign_creator.
Written out rather than imported so the probe can vary ``api`` and ``app_id``
independently -- figuring out which combination the gateway accepts is the
entire point of running it.
"""
content = api
if data is not None:
content += json.dumps(data, separators=(",", ":"), ensure_ascii=False)
digest = hashlib.md5(content.encode("utf-8")).hexdigest()
timestamp_ms = int(datetime.now().timestamp() * 1000)
plaintext = f"x1={digest};x2={XYW_ENV_FLAGS};x3={a1};x4={timestamp_ms};"
encoded = base64.b64encode(plaintext.encode("utf-8")).decode("utf-8")
envelope = {
"signSvn": "56",
"signType": "x2",
"appId": app_id,
"signVersion": "1",
"payload": _aes_encrypt_hex(encoded),
}
x_s = "XYW_" + base64.b64encode(
json.dumps(envelope, separators=(",", ":")).encode("utf-8")
).decode("utf-8")
return {"x-s": x_s, "x-t": str(timestamp_ms)}
def build_query(start_days_ago: int, end_days_ago: int, page_size: int = 10) -> str:
"""Query string exactly as the working collector builds it: epoch millis."""
today = datetime.now().replace(hour=0, minute=0, second=0, microsecond=0)
def ms(days_ago: int, at_end: bool) -> int:
day = (today - timedelta(days=days_ago)).date()
clock = time(23, 59, 59) if at_end else time(0, 0, 0)
return int(datetime.combine(day, clock).timestamp() * 1000)
return urllib.parse.urlencode(
{
"post_begin_time": ms(start_days_ago, False),
"post_end_time": ms(end_days_ago, True),
"type": 0,
"page_size": page_size,
"page_num": 1,
}
)
async def cookies_from_cdp() -> dict[str, str]:
"""Session cookies for creator.xiaohongshu.com, read out of the live browser."""
from playwright.async_api import async_playwright
playwright = await async_playwright().start()
try:
browser = await playwright.chromium.connect_over_cdp(
"http://127.0.0.1:9222", timeout=15000
)
jars = {}
for context in browser.contexts:
for cookie in await context.cookies():
jars[cookie["name"]] = cookie["value"]
return jars
finally:
await playwright.stop()
def classify(status: int, body: str) -> str:
if status == 406:
return "406 —— 网关拒了签名(回退浏览器拦截路线)"
if status == 200:
return "200 —— 通了,可以考虑解析数据"
if status in (401, 403):
return f"{status} —— 签名过了,只差创作者会话(好消息)"
return f"{status} —— 未知,需要看响应体"
async def main() -> int:
import httpx
ap = argparse.ArgumentParser()
ap.add_argument("--start-days-ago", type=int, default=30)
ap.add_argument("--end-days-ago", type=int, default=0)
ap.add_argument("--app-id", default="ugc", help="参考实现用 ugc;xhshow 默认 xhs-pc-web")
ap.add_argument("--cookie", default="", help="留空则从 CDP 浏览器读取")
ap.add_argument("--show-body", action="store_true", help="打印响应前 800 字符")
ap.add_argument(
"--no-cookie-header",
action="store_true",
help="签名照签(仍需 a1)但不发 cookie 头,用来分清"
"「空数据是缺会话」还是「接口本身就这样」",
)
args = ap.parse_args()
if args.cookie:
cookies = dict(
pair.split("=", 1) for pair in args.cookie.split("; ") if "=" in pair
)
else:
cookies = await cookies_from_cdp()
print(f" cookie 条数 {len(cookies)},名字: {sorted(cookies)}")
if not cookies.get("a1"):
print(" ✗ 没有 a1 —— 签名必须用它,无法继续")
return 2
query = build_query(args.start_days_ago, args.end_days_ago)
cookie_header = "; ".join(f"{k}={v}" for k, v in cookies.items())
headers_common = {
"user-agent": USER_AGENT,
"accept": "application/json, text/plain, */*",
"origin": CREATOR_ORIGIN,
"referer": f"{CREATOR_ORIGIN}/statistics/data-analysis",
"accept-language": "zh-CN,zh;q=0.9",
}
if not args.no_cookie_header:
headers_common["cookie"] = cookie_header
# Three signing variants: the reference bakes the query into the signed string,
# but the exact form is not documented beyond an example with no query at all.
variants = {
"path+q(参考实现写法)": f"url={NOTE_LIST_PATH}?{query}",
"path only": f"url={NOTE_LIST_PATH}",
"裸 path+query(无 url= 前缀)": f"{NOTE_LIST_PATH}?{query}",
}
url = f"{CREATOR_ORIGIN}{NOTE_LIST_PATH}?{query}"
async with httpx.AsyncClient(timeout=25, follow_redirects=False) as client:
for label, api in variants.items():
signature = sign_xyw(api, cookies["a1"], app_id=args.app_id)
headers = {**headers_common, **signature}
try:
response = await client.get(url, headers=headers)
except Exception as exc: # noqa: BLE001
print(f" [{label}] 请求异常: {exc.__class__.__name__}: {exc}")
continue
print(f"\n [{label}]")
print(f" HTTP {response.status_code} {classify(response.status_code, response.text)}")
body = response.text or ""
if body:
print(f" 响应前 160 字符: {body[:160]!r}")
if args.show_body and body:
print(f" 完整响应: {body[:800]}")
print("\n 提示:若三种都返回 406,再试 --app-id xhs-pc-web。")
return 0
if __name__ == "__main__":
import asyncio
sys.exit(asyncio.run(main()))
+144
View File
@@ -0,0 +1,144 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Phase 0, second half: read the real request off the real page.
The signed-request probe proved the signature is forgeable and that the main-site
cookie authenticates the creator backend -- but the endpoint it called returned an
envelope with no payload, which no real list endpoint does. That path came from a
third-party repo and may simply be stale.
So stop guessing at paths and watch the page. This opens the data-analysis page in
the browser that is already signed in, records every creator API call the page
itself makes, and prints each request's URL, method and the *shape* of the
response. The page's own requests are the ground truth.
Read-only: it navigates and observes. Nothing is submitted.
"""
import argparse
import asyncio
import json
import sys
from collections import Counter
DEFAULT_URL = "https://creator.xiaohongshu.com/statistics/data-analysis"
def shape(value, depth: int = 0) -> str:
"""Describe a JSON value's structure without dumping its data."""
if depth > 3:
return "…"
if isinstance(value, dict):
if not value:
return "{}"
inner = ", ".join(f"{k}: {shape(v, depth + 1)}" for k, v in list(value.items())[:12])
return "{" + inner + "}"
if isinstance(value, list):
if not value:
return "[]"
return f"[{len(value)} × {shape(value[0], depth + 1)}]"
if isinstance(value, str):
# Short values are shown as themselves -- the interesting ones here are
# status codes, roles and permission names, and "str(14)" tells you
# nothing. Long ones are almost always ids or urls, so only their length.
return repr(value) if len(value) <= 40 else f"str({len(value)})"
if isinstance(value, bool):
return str(value)
if isinstance(value, (int, float)):
return str(value)
return type(value).__name__
async def main() -> int:
from playwright.async_api import async_playwright
ap = argparse.ArgumentParser()
ap.add_argument("--url", default=DEFAULT_URL)
ap.add_argument("--wait", type=int, default=25, help="观察窗口(秒)")
ap.add_argument("--filter", default="/api/galaxy", help="只记录 URL 含此串的请求")
args = ap.parse_args()
playwright = await async_playwright().start()
seen: Counter[str] = Counter()
details: list[str] = []
try:
browser = await playwright.chromium.connect_over_cdp(
"http://127.0.0.1:9222", timeout=15000
)
context = browser.contexts[0]
async def on_response(response):
url = response.url
if args.filter not in url:
return
key = url.split("?")[0]
seen[key] += 1
if seen[key] > 1:
return
request = response.request
line = [f"\n {request.method} {key}"]
line.append(f" HTTP {response.status}")
query = url.split("?", 1)[1] if "?" in url else ""
if query:
line.append(f" 查询串: {query[:300]}")
post = request.post_data
if post:
line.append(f" POST body: {post[:300]}")
try:
payload = await response.json()
line.append(f" 响应结构: {shape(payload)[:600]}")
except Exception:
try:
text = await response.text()
line.append(f" 响应(非JSON)前 200: {text[:200]!r}")
except Exception as exc: # noqa: BLE001
line.append(f" 响应不可读: {exc.__class__.__name__}")
details.append("\n".join(line))
context.on("response", on_response)
page = await context.new_page()
try:
await page.goto(args.url, wait_until="domcontentloaded", timeout=45000)
print(f" 落地 URL: {page.url[:120]}")
title = await page.title()
print(f" 标题: {title[:80]!r}")
# A creator console that is genuinely reachable renders its shell; a
# login gate does not. This is the cheapest "are we in?" signal.
for label, selector in (
("登录表单", "//input[@type='password']"),
("扫码登录", "//*[contains(@class,'qrcode') or contains(@class,'qr-code')]"),
):
if await page.locator(selector).count() > 0:
print(f" ★ 页面上出现「{label}」—— 这个账号似乎没有创作者后台会话")
print(f"\n 观察 {args.wait} 秒,记录页面自己发的请求…")
await asyncio.sleep(args.wait)
finally:
try:
await page.close()
except Exception:
pass
context.remove_listener("response", on_response)
if not details:
print("\n ★ 没有捕获到任何匹配的请求 —— 页面很可能停在登录页,没有发出数据请求")
for block in details:
print(block)
print(f"\n 捕获到的接口(去重): {len(seen)}")
for key, count in seen.most_common():
print(f" ×{count} {key}")
finally:
await playwright.stop()
return 0
if __name__ == "__main__":
sys.exit(asyncio.run(main()))