Files
butubb c2b310c7bf
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
feat(creator): 新增「运营」模块 —— 多账号扫码登录与创作者后台数据
侧边栏在「监控」右边加了「运营」:账号列表 → 点进二级详情看该账号的数据。

【为什么是独立模块而不是监控的子视图】两者形状不同:监控是公开数据(点赞/收藏/评论/分享)的每轮快照+差分;运营是创作者后台按日期给出的曝光/观看/完播率/涨粉。凭据不同、采集方式也不同 —— 那边要浏览器登录态,这边是纯请求。硬塞进同一个模型会同时污染两边。

【扫码登录的关键差异】监控的扫码把登录态写进浏览器默认 profile(爬虫要复用)。运营要的是 cookie 字符串(纯请求够用),所以每次登录开一个**临时上下文**,扫完取出 cookie 就丢弃 —— 登第二个账号不会把第一个顶掉,也不影响监控那个登录态,十个账号互不干扰。

【决策依据】tools/probe_creator_api.py 的 Phase 0 实测:签名可自造(XYW_:MD5 → base64 → AES-128-CBC,与 xhshow 内置实现常量逐字节一致);主站 cookie 即可认证创作者后台;接口与参数已与真实页面对齐。

后端:
- api/creator/models.py: creator_account / creator_note_stat。**复用 MonitorBase**,这样 create_all 与上一轮改成元数据驱动的 _ensure_columns 会自动覆盖新表
- api/creator/signing.py: XYW_ 签名,带三条实测结论(url= 前缀、appId=ugc、401 与 406 的区别)
- api/creator/client.py: 纯 httpx 客户端。字段名尚未亲眼验证过,所以写成多别名匹配;解析不出来存 None 而非 0
- api/creator/service.py: 账号 CRUD 与同步。cookie 绝不进入对外结构,只给 has_cookie
- api/creator/login.py: 临时上下文的扫码登录
- api/routers/creator.py: 8 条路由,全部带鉴权

前端:
- 侧边栏「运营」+ OperationView(账号列表 → 二级详情)+ AddAccountDialog
- 权限状态显眼呈现:pending 时照抄后台原话「已为您申请数据权限,次日可查看」,并说明此时同步返回 0 条是正常的,不是采集失败

测试:tests/test_creator_client.py 新增 48 例,含「cookie 不得出现在对外结构里」这条不变量,以及权限未生效时空壳响应的处理。
2026-10-07 16:30:45 +08:00

322 lines
12 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/main.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""
综合采集平台 API Server
Start command: uvicorn api.main:app --port 8080 --reload
Or: python -m api.main
"""
import asyncio
import os
import sys
import subprocess
from contextlib import asynccontextmanager
from pathlib import Path
# Project root directory (used for running subprocesses like uv run main.py)
PROJECT_ROOT = Path(__file__).parent.parent
# Load .env before importing anything that reads os.getenv at module import time
# (config/db_config.py does). python-dotenv was already a declared dependency but
# nothing ever called it, so the shipped .env.example had no effect.
from dotenv import load_dotenv
load_dotenv(PROJECT_ROOT / ".env")
import uvicorn
from fastapi import Depends, FastAPI
from fastapi.middleware.cors import CORSMiddleware
from fastapi.staticfiles import StaticFiles
from fastapi.responses import FileResponse
from .auth import ensure_initial_credential, require_auth
from .routers import (
auth_router,
crawler_router,
creator_router,
data_router,
monitor_router,
settings_router,
websocket_router,
)
from .services.interpreter import describe_interpreter, resolve_python_cmd
@asynccontextmanager
async def lifespan(_app: FastAPI):
"""Start the monitor scheduler with the server, and shut it down cleanly.
The scheduler is a background asyncio task, so it must not be tied to a
browser session the way the log broadcaster is -- a scheduled run has to
happen whether or not anyone has the UI open.
"""
from .creator.login import shutdown as shutdown_creator_login
from .monitor.db import dispose_engine, init_db
from .monitor.qrlogin import shutdown as shutdown_qrlogin
from .monitor.scheduler import monitor_scheduler
await init_db()
# The WebUI bundle is gitignored and built separately, so a deployment that
# forgot it would otherwise come up looking healthy and serve a bare JSON
# stub at "/" -- worth one loud line at boot rather than a puzzled operator.
if not os.path.exists(os.path.join(WEBUI_DIR, "index.html")):
print(
"[综合采集平台] 警告:未找到前端产物 api/webui/index.html,"
"根路径只会返回一段 JSON。请先在 webui/ 下执行 npm run build。",
flush=True,
)
generated = await ensure_initial_credential()
if generated:
# Printed once, on the run that creates it. There is no unauthenticated
# "set your password" endpoint on purpose: on a LAN bind that would be a
# claim-the-instance race.
rule = "=" * 68
print(
f"\n{rule}\n"
" WebUI 首次启动,已生成登录密码:\n"
f"\n {generated}\n"
"\n 请立即登录并修改。忘记密码时可设置环境变量 MC_PASSWORD 后重启。\n"
f"{rule}\n",
flush=True,
)
await monitor_scheduler.start()
try:
yield
finally:
await monitor_scheduler.stop()
# Drops the tab a QR login may have opened and stops the Playwright
# client; leaving them would strand a driver process on every restart.
await shutdown_qrlogin()
# Same for the operator's account logins, which run in throwaway browser
# contexts -- those would otherwise be left open in the operator's Chrome.
await shutdown_creator_login()
await dispose_engine()
# Docs are disabled deliberately: /docs, /redoc and /openapi.json are
# unauthenticated by default, which would hand out a complete map of the API
# (and a "Try it out" console that 401s anyway).
app = FastAPI(
title="综合采集平台 API",
description="API for controlling 综合采集平台 from WebUI",
version="1.0.0",
lifespan=lifespan,
docs_url=None,
redoc_url=None,
openapi_url=None,
)
# Get webui static files directory
WEBUI_DIR = os.path.join(os.path.dirname(__file__), "webui")
# CORS only matters for a split-origin setup. In production this app serves the
# SPA itself, and in development Vite proxies /api here (see webui/vite.config.ts),
# so the browser always sees a single origin and CORS never actually triggers.
# Kept as an explicit allowlist -- never "*", which is invalid next to
# allow_credentials -- and extensible via env for a dev server reached over LAN.
_extra_origins = [o.strip() for o in os.getenv("MC_CORS_ORIGINS", "").split(",") if o.strip()]
app.add_middleware(
CORSMiddleware,
allow_origins=[
"http://localhost:5173", # Vite dev server
"http://localhost:3000", # Backup port
"http://127.0.0.1:5173",
"http://127.0.0.1:3000",
*_extra_origins,
],
allow_origin_regex=os.getenv("MC_CORS_ORIGIN_REGEX") or None,
allow_credentials=True,
allow_methods=["*"],
allow_headers=["*"],
)
# Register routers.
# The auth router stays open -- it is the way in. Everything else under /api
# requires a session. Enforcement is a Depends applied per router rather than
# app-wide middleware, because middleware needs a hand-rolled path allowlist and,
# more importantly, never sees WebSocket scopes at all.
app.include_router(auth_router, prefix="/api")
app.include_router(crawler_router, prefix="/api", dependencies=[Depends(require_auth)])
app.include_router(creator_router, prefix="/api", dependencies=[Depends(require_auth)])
app.include_router(data_router, prefix="/api", dependencies=[Depends(require_auth)])
app.include_router(monitor_router, prefix="/api", dependencies=[Depends(require_auth)])
app.include_router(settings_router, prefix="/api", dependencies=[Depends(require_auth)])
app.include_router(websocket_router, prefix="/api")
@app.get("/")
async def serve_frontend():
"""Return frontend page"""
index_path = os.path.join(WEBUI_DIR, "index.html")
if os.path.exists(index_path):
return FileResponse(index_path)
return {
"message": "综合采集平台 API",
"version": "1.0.0",
"note": "WebUI not found, please build it first: cd webui && npm run build"
}
@app.get("/api/health")
async def health_check():
return {"status": "ok"}
@app.get("/api/env/check", dependencies=[Depends(require_auth)])
async def check_environment():
"""Check whether the crawler environment is configured correctly"""
try:
# Run `main.py --help` to check the environment.
# Resolve the interpreter the same way the crawler manager does, so this
# check can never disagree with how main.py is actually executed.
# Use PROJECT_ROOT so it works regardless of where uvicorn was started.
python_cmd = resolve_python_cmd()
if sys.platform == "win32":
loop = asyncio.get_running_loop()
process = await loop.run_in_executor(
None,
lambda: subprocess.run(
[*python_cmd, "main.py", "--help"],
capture_output=True,
timeout=30.0,
cwd=str(PROJECT_ROOT)
)
)
stdout, stderr = process.stdout, process.stderr # bytes
else:
process = await asyncio.create_subprocess_exec(
*python_cmd, "main.py", "--help",
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
cwd=str(PROJECT_ROOT) # Project root directory
)
stdout, stderr = await asyncio.wait_for(
process.communicate(),
timeout=30.0 # 30 seconds timeout
)
if process.returncode == 0:
return {
"success": True,
"message": "环境配置正确",
"interpreter": describe_interpreter(),
"output": stdout.decode("utf-8", errors="ignore")[:500] # Truncate to first 500 characters
}
else:
error_msg = stderr.decode("utf-8", errors="ignore") or stdout.decode("utf-8", errors="ignore")
return {
"success": False,
"message": "Environment check failed",
"error": error_msg[:500]
}
except asyncio.TimeoutError:
return {
"success": False,
"message": "Environment check timeout",
"error": "Command execution exceeded 30 seconds"
}
except FileNotFoundError:
return {
"success": False,
"message": "Python interpreter not found",
"error": (
"Neither uv nor a usable interpreter was found. Install uv, or create a "
"project virtualenv (.venv) with the requirements installed."
)
}
except Exception as e:
return {
"success": False,
"message": "Environment check error",
"error": f"{type(e).__name__}: {str(e) or 'Unknown'}"
}
@app.get("/api/config/platforms", dependencies=[Depends(require_auth)])
async def get_platforms():
"""Platform capability matrix.
Returns what each platform's crawler supports (modes, metrics, comment
levels, media) *and* whether the monitoring layer has been wired up for it.
The UI renders its platform switcher and metric columns from this, so the
two are never allowed to drift apart.
"""
from .monitor.platforms import describe_all
return {"platforms": describe_all()}
@app.get("/api/config/options", dependencies=[Depends(require_auth)])
async def get_config_options():
"""Get all configuration options"""
return {
"login_types": [
{"value": "qrcode", "label": "扫码登录"},
# Named for what it now does: the value itself is no longer typed
# here, it is reused from Settings.
{"value": "cookie", "label": "复用已保存的 Cookie"},
],
"crawler_types": [
{"value": "search", "label": "Search Mode"},
{"value": "detail", "label": "Detail Mode"},
{"value": "creator", "label": "Creator Mode"},
],
"save_options": [
{"value": "jsonl", "label": "JSONL File"},
{"value": "json", "label": "JSON File"},
{"value": "csv", "label": "CSV File"},
{"value": "excel", "label": "Excel File"},
{"value": "sqlite", "label": "SQLite Database"},
{"value": "db", "label": "MySQL Database"},
{"value": "mongodb", "label": "MongoDB Database"},
],
}
# Mount static resources - must be placed after all routes
if os.path.exists(WEBUI_DIR):
assets_dir = os.path.join(WEBUI_DIR, "assets")
if os.path.exists(assets_dir):
app.mount("/assets", StaticFiles(directory=assets_dir), name="assets")
# Mount logos directory
logos_dir = os.path.join(WEBUI_DIR, "logos")
if os.path.exists(logos_dir):
app.mount("/logos", StaticFiles(directory=logos_dir), name="logos")
# Mount other static files (e.g., vite.svg)
app.mount("/static", StaticFiles(directory=WEBUI_DIR), name="webui-static")
if __name__ == "__main__":
# Loopback by default: the safe choice for anyone who has not thought about
# exposure. Set MC_HOST=0.0.0.0 (e.g. in .env) for LAN access. Before this,
# `python -m api.main` bound 0.0.0.0 while the documented `uvicorn api.main:app`
# bound loopback -- two launch paths with different exposure.
host = os.getenv("MC_HOST", "127.0.0.1")
port = int(os.getenv("MC_PORT", "8080"))
if host not in ("127.0.0.1", "localhost", "::1"):
print(
f"[综合采集平台] 监听 {host}:{port},局域网内其他机器可访问。\n"
f"[综合采集平台] 已启用密码鉴权;如需暴露到可信网络之外,请走 HTTPS 反向代理。",
flush=True,
)
uvicorn.run(app, host=host, port=port)