Files
MediaCrawler/tools/probe_creator_page.py
T
butubb 2613f7577f
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
feat(creator): Phase 0 探针 —— 创作者后台数据可以纯请求拿到
结论:签名可自造、主站 cookie 即可认证、接口与参数已与真实页面对齐。不需要浏览器、不需要独立的创作者登录。

- tools/probe_creator_api.py: 纯 HTTP 探针。用 XYW_ 算法自签(MD5 → base64 → AES-128-CBC,
  密钥与 IV 与 xhshow/config/config.py 逐字节一致),对 note/analyze/list 发请求
- tools/probe_creator_page.py: 打开真实数据分析页,记录页面自己发的请求,作为地面真相

Phase 0 的三条实测结论:
1. 签名可伪造。三种写法里只有「url= + 路径 + 查询串」被接受(200);仅路径、或裸路径都 406。
   并且不带 cookie 时返回的是应用层 401「无登录信息」而非网关 406 —— 说明签名每次都已通过
2. 主站 .xiaohongshu.com 的 cookie 就能认证创作者后台,不需要单独的创作者会话
3. 当前账号 dfg 返回空数据不是技术问题:permission/query 的 tip_msg 是
   「已为您申请数据权限,次日可查看」,display/status 均为 0,即权限尚未生效

关键佐证:真实页面调 note/analyze/list 用的查询串与本探针生成的完全一致,且拿到同一份空响应。
2026-10-07 16:22:15 +08:00

145 lines
5.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Phase 0, second half: read the real request off the real page.
The signed-request probe proved the signature is forgeable and that the main-site
cookie authenticates the creator backend -- but the endpoint it called returned an
envelope with no payload, which no real list endpoint does. That path came from a
third-party repo and may simply be stale.
So stop guessing at paths and watch the page. This opens the data-analysis page in
the browser that is already signed in, records every creator API call the page
itself makes, and prints each request's URL, method and the *shape* of the
response. The page's own requests are the ground truth.
Read-only: it navigates and observes. Nothing is submitted.
"""
import argparse
import asyncio
import json
import sys
from collections import Counter
DEFAULT_URL = "https://creator.xiaohongshu.com/statistics/data-analysis"
def shape(value, depth: int = 0) -> str:
"""Describe a JSON value's structure without dumping its data."""
if depth > 3:
return "…"
if isinstance(value, dict):
if not value:
return "{}"
inner = ", ".join(f"{k}: {shape(v, depth + 1)}" for k, v in list(value.items())[:12])
return "{" + inner + "}"
if isinstance(value, list):
if not value:
return "[]"
return f"[{len(value)} × {shape(value[0], depth + 1)}]"
if isinstance(value, str):
# Short values are shown as themselves -- the interesting ones here are
# status codes, roles and permission names, and "str(14)" tells you
# nothing. Long ones are almost always ids or urls, so only their length.
return repr(value) if len(value) <= 40 else f"str({len(value)})"
if isinstance(value, bool):
return str(value)
if isinstance(value, (int, float)):
return str(value)
return type(value).__name__
async def main() -> int:
from playwright.async_api import async_playwright
ap = argparse.ArgumentParser()
ap.add_argument("--url", default=DEFAULT_URL)
ap.add_argument("--wait", type=int, default=25, help="观察窗口(秒)")
ap.add_argument("--filter", default="/api/galaxy", help="只记录 URL 含此串的请求")
args = ap.parse_args()
playwright = await async_playwright().start()
seen: Counter[str] = Counter()
details: list[str] = []
try:
browser = await playwright.chromium.connect_over_cdp(
"http://127.0.0.1:9222", timeout=15000
)
context = browser.contexts[0]
async def on_response(response):
url = response.url
if args.filter not in url:
return
key = url.split("?")[0]
seen[key] += 1
if seen[key] > 1:
return
request = response.request
line = [f"\n {request.method} {key}"]
line.append(f" HTTP {response.status}")
query = url.split("?", 1)[1] if "?" in url else ""
if query:
line.append(f" 查询串: {query[:300]}")
post = request.post_data
if post:
line.append(f" POST body: {post[:300]}")
try:
payload = await response.json()
line.append(f" 响应结构: {shape(payload)[:600]}")
except Exception:
try:
text = await response.text()
line.append(f" 响应(非JSON)前 200: {text[:200]!r}")
except Exception as exc: # noqa: BLE001
line.append(f" 响应不可读: {exc.__class__.__name__}")
details.append("\n".join(line))
context.on("response", on_response)
page = await context.new_page()
try:
await page.goto(args.url, wait_until="domcontentloaded", timeout=45000)
print(f" 落地 URL: {page.url[:120]}")
title = await page.title()
print(f" 标题: {title[:80]!r}")
# A creator console that is genuinely reachable renders its shell; a
# login gate does not. This is the cheapest "are we in?" signal.
for label, selector in (
("登录表单", "//input[@type='password']"),
("扫码登录", "//*[contains(@class,'qrcode') or contains(@class,'qr-code')]"),
):
if await page.locator(selector).count() > 0:
print(f" ★ 页面上出现「{label}」—— 这个账号似乎没有创作者后台会话")
print(f"\n 观察 {args.wait} 秒,记录页面自己发的请求…")
await asyncio.sleep(args.wait)
finally:
try:
await page.close()
except Exception:
pass
context.remove_listener("response", on_response)
if not details:
print("\n ★ 没有捕获到任何匹配的请求 —— 页面很可能停在登录页,没有发出数据请求")
for block in details:
print(block)
print(f"\n 捕获到的接口(去重): {len(seen)}")
for key, count in seen.most_common():
print(f" ×{count} {key}")
finally:
await playwright.stop()
return 0
if __name__ == "__main__":
sys.exit(asyncio.run(main()))