# -*- coding: utf-8 -*- # Copyright (c) 2025 relakkes@gmail.com # # This file is part of MediaCrawler project. # Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/main.py # GitHub: https://github.com/NanmiCoder # Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 # # 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: # 1. 不得用于任何商业用途。 # 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 # 3. 不得进行大规模爬取或对平台造成运营干扰。 # 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 # 5. 不得用于任何非法或不当的用途。 # # 详细许可条款请参阅项目根目录下的LICENSE文件。 # 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 """ 综合采集平台 API Server Start command: uvicorn api.main:app --port 8080 --reload Or: python -m api.main """ import asyncio import os import sys import subprocess from contextlib import asynccontextmanager from pathlib import Path # Project root directory (used for running subprocesses like uv run main.py) PROJECT_ROOT = Path(__file__).parent.parent # Load .env before importing anything that reads os.getenv at module import time # (config/db_config.py does). python-dotenv was already a declared dependency but # nothing ever called it, so the shipped .env.example had no effect. from dotenv import load_dotenv load_dotenv(PROJECT_ROOT / ".env") import uvicorn from fastapi import Depends, FastAPI from fastapi.middleware.cors import CORSMiddleware from fastapi.staticfiles import StaticFiles from fastapi.responses import FileResponse from .auth import ensure_initial_credential, require_auth from .routers import ( auth_router, crawler_router, data_router, monitor_router, settings_router, websocket_router, ) from .services.interpreter import describe_interpreter, resolve_python_cmd @asynccontextmanager async def lifespan(_app: FastAPI): """Start the monitor scheduler with the server, and shut it down cleanly. The scheduler is a background asyncio task, so it must not be tied to a browser session the way the log broadcaster is -- a scheduled run has to happen whether or not anyone has the UI open. """ from .monitor.db import dispose_engine, init_db from .monitor.qrlogin import shutdown as shutdown_qrlogin from .monitor.scheduler import monitor_scheduler await init_db() generated = await ensure_initial_credential() if generated: # Printed once, on the run that creates it. There is no unauthenticated # "set your password" endpoint on purpose: on a LAN bind that would be a # claim-the-instance race. rule = "=" * 68 print( f"\n{rule}\n" " WebUI 首次启动,已生成登录密码:\n" f"\n {generated}\n" "\n 请立即登录并修改。忘记密码时可设置环境变量 MC_PASSWORD 后重启。\n" f"{rule}\n", flush=True, ) await monitor_scheduler.start() try: yield finally: await monitor_scheduler.stop() # Drops the tab a QR login may have opened and stops the Playwright # client; leaving them would strand a driver process on every restart. await shutdown_qrlogin() await dispose_engine() # Docs are disabled deliberately: /docs, /redoc and /openapi.json are # unauthenticated by default, which would hand out a complete map of the API # (and a "Try it out" console that 401s anyway). app = FastAPI( title="综合采集平台 API", description="API for controlling 综合采集平台 from WebUI", version="1.0.0", lifespan=lifespan, docs_url=None, redoc_url=None, openapi_url=None, ) # Get webui static files directory WEBUI_DIR = os.path.join(os.path.dirname(__file__), "webui") # CORS only matters for a split-origin setup. In production this app serves the # SPA itself, and in development Vite proxies /api here (see webui/vite.config.ts), # so the browser always sees a single origin and CORS never actually triggers. # Kept as an explicit allowlist -- never "*", which is invalid next to # allow_credentials -- and extensible via env for a dev server reached over LAN. _extra_origins = [o.strip() for o in os.getenv("MC_CORS_ORIGINS", "").split(",") if o.strip()] app.add_middleware( CORSMiddleware, allow_origins=[ "http://localhost:5173", # Vite dev server "http://localhost:3000", # Backup port "http://127.0.0.1:5173", "http://127.0.0.1:3000", *_extra_origins, ], allow_origin_regex=os.getenv("MC_CORS_ORIGIN_REGEX") or None, allow_credentials=True, allow_methods=["*"], allow_headers=["*"], ) # Register routers. # The auth router stays open -- it is the way in. Everything else under /api # requires a session. Enforcement is a Depends applied per router rather than # app-wide middleware, because middleware needs a hand-rolled path allowlist and, # more importantly, never sees WebSocket scopes at all. app.include_router(auth_router, prefix="/api") app.include_router(crawler_router, prefix="/api", dependencies=[Depends(require_auth)]) app.include_router(data_router, prefix="/api", dependencies=[Depends(require_auth)]) app.include_router(monitor_router, prefix="/api", dependencies=[Depends(require_auth)]) app.include_router(settings_router, prefix="/api", dependencies=[Depends(require_auth)]) app.include_router(websocket_router, prefix="/api") @app.get("/") async def serve_frontend(): """Return frontend page""" index_path = os.path.join(WEBUI_DIR, "index.html") if os.path.exists(index_path): return FileResponse(index_path) return { "message": "综合采集平台 API", "version": "1.0.0", "note": "WebUI not found, please build it first: cd webui && npm run build" } @app.get("/api/health") async def health_check(): return {"status": "ok"} @app.get("/api/env/check", dependencies=[Depends(require_auth)]) async def check_environment(): """Check whether the crawler environment is configured correctly""" try: # Run `main.py --help` to check the environment. # Resolve the interpreter the same way the crawler manager does, so this # check can never disagree with how main.py is actually executed. # Use PROJECT_ROOT so it works regardless of where uvicorn was started. python_cmd = resolve_python_cmd() if sys.platform == "win32": loop = asyncio.get_running_loop() process = await loop.run_in_executor( None, lambda: subprocess.run( [*python_cmd, "main.py", "--help"], capture_output=True, timeout=30.0, cwd=str(PROJECT_ROOT) ) ) stdout, stderr = process.stdout, process.stderr # bytes else: process = await asyncio.create_subprocess_exec( *python_cmd, "main.py", "--help", stdout=subprocess.PIPE, stderr=subprocess.PIPE, cwd=str(PROJECT_ROOT) # Project root directory ) stdout, stderr = await asyncio.wait_for( process.communicate(), timeout=30.0 # 30 seconds timeout ) if process.returncode == 0: return { "success": True, "message": "环境配置正确", "interpreter": describe_interpreter(), "output": stdout.decode("utf-8", errors="ignore")[:500] # Truncate to first 500 characters } else: error_msg = stderr.decode("utf-8", errors="ignore") or stdout.decode("utf-8", errors="ignore") return { "success": False, "message": "Environment check failed", "error": error_msg[:500] } except asyncio.TimeoutError: return { "success": False, "message": "Environment check timeout", "error": "Command execution exceeded 30 seconds" } except FileNotFoundError: return { "success": False, "message": "Python interpreter not found", "error": ( "Neither uv nor a usable interpreter was found. Install uv, or create a " "project virtualenv (.venv) with the requirements installed." ) } except Exception as e: return { "success": False, "message": "Environment check error", "error": f"{type(e).__name__}: {str(e) or 'Unknown'}" } @app.get("/api/config/platforms", dependencies=[Depends(require_auth)]) async def get_platforms(): """Platform capability matrix. Returns what each platform's crawler supports (modes, metrics, comment levels, media) *and* whether the monitoring layer has been wired up for it. The UI renders its platform switcher and metric columns from this, so the two are never allowed to drift apart. """ from .monitor.platforms import describe_all return {"platforms": describe_all()} @app.get("/api/config/options", dependencies=[Depends(require_auth)]) async def get_config_options(): """Get all configuration options""" return { "login_types": [ {"value": "qrcode", "label": "扫码登录"}, # Named for what it now does: the value itself is no longer typed # here, it is reused from Settings. {"value": "cookie", "label": "复用已保存的 Cookie"}, ], "crawler_types": [ {"value": "search", "label": "Search Mode"}, {"value": "detail", "label": "Detail Mode"}, {"value": "creator", "label": "Creator Mode"}, ], "save_options": [ {"value": "jsonl", "label": "JSONL File"}, {"value": "json", "label": "JSON File"}, {"value": "csv", "label": "CSV File"}, {"value": "excel", "label": "Excel File"}, {"value": "sqlite", "label": "SQLite Database"}, {"value": "db", "label": "MySQL Database"}, {"value": "mongodb", "label": "MongoDB Database"}, ], } # Mount static resources - must be placed after all routes if os.path.exists(WEBUI_DIR): assets_dir = os.path.join(WEBUI_DIR, "assets") if os.path.exists(assets_dir): app.mount("/assets", StaticFiles(directory=assets_dir), name="assets") # Mount logos directory logos_dir = os.path.join(WEBUI_DIR, "logos") if os.path.exists(logos_dir): app.mount("/logos", StaticFiles(directory=logos_dir), name="logos") # Mount other static files (e.g., vite.svg) app.mount("/static", StaticFiles(directory=WEBUI_DIR), name="webui-static") if __name__ == "__main__": # Loopback by default: the safe choice for anyone who has not thought about # exposure. Set MC_HOST=0.0.0.0 (e.g. in .env) for LAN access. Before this, # `python -m api.main` bound 0.0.0.0 while the documented `uvicorn api.main:app` # bound loopback -- two launch paths with different exposure. host = os.getenv("MC_HOST", "127.0.0.1") port = int(os.getenv("MC_PORT", "8080")) if host not in ("127.0.0.1", "localhost", "::1"): print( f"[综合采集平台] 监听 {host}:{port},局域网内其他机器可访问。\n" f"[综合采集平台] 已启用密码鉴权;如需暴露到可信网络之外,请走 HTTPS 反向代理。", flush=True, ) uvicorn.run(app, host=host, port=port)