feat: 监控面板 / 登录鉴权 / 多平台切换 / MySQL
在上游 MediaCrawler 之上新增一层: - 监控层 api/monitor/ —— 多博主/多笔记的定时采集、指标快照差分、报表、 企业微信通知。每轮采集写入独立目录,差分才成立。 - WebUI 登录鉴权 api/auth.py —— PBKDF2 口令 + 服务端会话,/api 全接口防护。 WebSocket 单独加依赖:BaseHTTPMiddleware 对 ws 作用域直接放行,覆盖不到。 - 全局平台切换 + 能力矩阵 —— 如实区分「爬虫模块支持」与「监控层已接线」, 未接通的平台直接拒绝建任务,而不是静默跑空。 - 监控库改用 MySQL 5.7(可回退 SQLite 供测试):逐表强制 utf8mb4 (服务端与库默认都是 latin1),启动校验所连 schema 以防写错库, 连接池 recycle + pre_ping 应对 MySQL 的 8 小时空闲断连。 修复上游缺陷: - xhs/core.py: 主页抓取失败会跳掉整个博主,导致一条作品都抓不到, 而那份资料只喂给一个空函数。改为尽力而为,失败不中断。 - xhs/login.py: cookie 登录只注入 web_session,冷启动签名会失败。 新增 INJECT_ALL_COOKIES 开关(默认关闭,原有行为不变)。 - requirements.txt: 补上 websockets。它在上游 pyproject.toml 里有声明、 这里漏了,导致 uvicorn 没有 WebSocket 能力,实时日志流从未工作。 改动过的上游文件清单及合并方式见 UPSTREAM.md。 测试:492 passed(另有 1 个既有的 Windows/gbk 上游测试失败,与本改动无关)
This commit is contained in:
+103
-10
@@ -25,6 +25,7 @@ from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
from ..schemas import CrawlerStartRequest, LogEntry
|
||||
from .interpreter import resolve_python_cmd
|
||||
|
||||
|
||||
class CrawlerManager:
|
||||
@@ -43,6 +44,11 @@ class CrawlerManager:
|
||||
self._project_root = Path(__file__).parent.parent.parent
|
||||
# Log queue - for pushing to WebSocket
|
||||
self._log_queue: Optional[asyncio.Queue] = None
|
||||
# Completion signalling for run_and_wait(). Polling `status` is unreliable
|
||||
# because stop() also resets it to "idle", and `self.process` gets replaced
|
||||
# by any concurrent start(), so waiters need an explicit event instead.
|
||||
self._done: asyncio.Event = asyncio.Event()
|
||||
self.last_exit_code: Optional[int] = None
|
||||
|
||||
@property
|
||||
def logs(self) -> List[LogEntry]:
|
||||
@@ -54,6 +60,43 @@ class CrawlerManager:
|
||||
self._log_queue = asyncio.Queue()
|
||||
return self._log_queue
|
||||
|
||||
def is_busy(self) -> bool:
|
||||
"""Whether a crawler process is currently alive.
|
||||
|
||||
This is the authoritative busy check -- `status` is a lagging indicator
|
||||
that manual stop() also resets.
|
||||
"""
|
||||
return self.process is not None and self.process.poll() is None
|
||||
|
||||
async def run_and_wait(
|
||||
self,
|
||||
config: CrawlerStartRequest,
|
||||
extra_args: Optional[List[str]] = None,
|
||||
timeout: Optional[float] = None,
|
||||
) -> int:
|
||||
"""Start a crawler run and block until it exits, returning the exit code.
|
||||
|
||||
Used by the monitor scheduler. Returns a negative value if the run was
|
||||
killed by `timeout` or if the process could not be started at all.
|
||||
"""
|
||||
started = await self.start(config, extra_args=extra_args)
|
||||
if not started:
|
||||
return -1
|
||||
|
||||
# Capture the process we just launched: a concurrent start() would
|
||||
# replace self.process, so poll this reference rather than the attribute.
|
||||
proc = self.process
|
||||
if proc is None:
|
||||
return -1
|
||||
|
||||
try:
|
||||
await asyncio.wait_for(self._done.wait(), timeout=timeout)
|
||||
except asyncio.TimeoutError:
|
||||
await self.stop()
|
||||
return -1
|
||||
|
||||
return self.last_exit_code if self.last_exit_code is not None else -1
|
||||
|
||||
def _create_log_entry(self, message: str, level: str = "info") -> LogEntry:
|
||||
"""Create log entry"""
|
||||
self._log_id += 1
|
||||
@@ -90,7 +133,11 @@ class CrawlerManager:
|
||||
return "debug"
|
||||
return "info"
|
||||
|
||||
async def start(self, config: CrawlerStartRequest) -> bool:
|
||||
async def start(
|
||||
self,
|
||||
config: CrawlerStartRequest,
|
||||
extra_args: Optional[List[str]] = None,
|
||||
) -> bool:
|
||||
"""Start crawler process"""
|
||||
async with self._lock:
|
||||
if self.process and self.process.poll() is None:
|
||||
@@ -99,6 +146,9 @@ class CrawlerManager:
|
||||
# Clear old logs
|
||||
self._logs = []
|
||||
self._log_id = 0
|
||||
# Reset completion signalling for this run
|
||||
self._done.clear()
|
||||
self.last_exit_code = None
|
||||
|
||||
# Clear pending queue (don't replace object to avoid WebSocket broadcast coroutine holding old queue reference)
|
||||
if self._log_queue is None:
|
||||
@@ -111,7 +161,7 @@ class CrawlerManager:
|
||||
pass
|
||||
|
||||
# Build command line arguments
|
||||
cmd = self._build_command(config)
|
||||
cmd = self._build_command(config, extra_args=extra_args)
|
||||
|
||||
# Log start information
|
||||
entry = self._create_log_entry(f"Starting crawler: {' '.join(cmd)}", "info")
|
||||
@@ -202,9 +252,13 @@ class CrawlerManager:
|
||||
"error_message": None
|
||||
}
|
||||
|
||||
def _build_command(self, config: CrawlerStartRequest) -> list:
|
||||
def _build_command(
|
||||
self,
|
||||
config: CrawlerStartRequest,
|
||||
extra_args: Optional[List[str]] = None,
|
||||
) -> list:
|
||||
"""Build main.py command line arguments"""
|
||||
cmd = ["uv", "run", "python", "main.py"]
|
||||
cmd = [*resolve_python_cmd(), "main.py"]
|
||||
|
||||
cmd.extend(["--platform", config.platform.value])
|
||||
cmd.extend(["--lt", config.login_type.value])
|
||||
@@ -232,22 +286,56 @@ class CrawlerManager:
|
||||
if config.max_comments_count is not None:
|
||||
cmd.extend(["--max_comments_count_singlenotes", str(config.max_comments_count)])
|
||||
|
||||
if config.cookies:
|
||||
# Each of these is only appended when explicitly set, so manual runs from
|
||||
# the Crawl tab keep exactly their previous behaviour.
|
||||
if config.save_data_path:
|
||||
cmd.extend(["--save_data_path", config.save_data_path])
|
||||
if config.enable_cdp_mode is not None:
|
||||
cmd.extend(["--enable_cdp_mode", "true" if config.enable_cdp_mode else "false"])
|
||||
if config.inject_all_cookies is not None:
|
||||
cmd.extend(["--inject_all_cookies", "true" if config.inject_all_cookies else "false"])
|
||||
if config.save_login_state is not None:
|
||||
cmd.extend(["--save_login_state", "true" if config.save_login_state else "false"])
|
||||
if config.max_concurrency_num is not None:
|
||||
cmd.extend(["--max_concurrency_num", str(config.max_concurrency_num)])
|
||||
if config.crawler_max_sleep_sec is not None:
|
||||
cmd.extend(["--crawler_max_sleep_sec", str(config.crawler_max_sleep_sec)])
|
||||
if config.enable_ip_proxy is not None:
|
||||
cmd.extend(["--enable_ip_proxy", "true" if config.enable_ip_proxy else "false"])
|
||||
if config.ip_proxy_pool_count is not None:
|
||||
cmd.extend(["--ip_proxy_pool_count", str(config.ip_proxy_pool_count)])
|
||||
if config.ip_proxy_provider_name:
|
||||
cmd.extend(["--ip_proxy_provider_name", config.ip_proxy_provider_name])
|
||||
if config.static_proxy_url:
|
||||
cmd.extend(["--static_proxy_url", config.static_proxy_url])
|
||||
|
||||
# Prefer a cookie file over passing the cookie on the command line, where
|
||||
# it would be visible in the process list.
|
||||
if config.cookies_file:
|
||||
cmd.extend(["--cookies_file", config.cookies_file])
|
||||
elif config.cookies:
|
||||
cmd.extend(["--cookies", config.cookies])
|
||||
|
||||
cmd.extend(["--headless", "true" if config.headless else "false"])
|
||||
|
||||
if extra_args:
|
||||
cmd.extend(extra_args)
|
||||
|
||||
return cmd
|
||||
|
||||
async def _read_output(self):
|
||||
"""Asynchronously read process output"""
|
||||
loop = asyncio.get_event_loop()
|
||||
# Capture the process this reader was started for. self.process can be
|
||||
# replaced by a subsequent start(), which would otherwise make us read
|
||||
# the exit code of the wrong run.
|
||||
proc = self.process
|
||||
|
||||
try:
|
||||
while self.process and self.process.poll() is None:
|
||||
while proc and proc.poll() is None:
|
||||
# Read a line in thread pool
|
||||
line = await loop.run_in_executor(
|
||||
None, self.process.stdout.readline
|
||||
None, proc.stdout.readline
|
||||
)
|
||||
if line:
|
||||
line = line.strip()
|
||||
@@ -257,9 +345,9 @@ class CrawlerManager:
|
||||
await self._push_log(entry)
|
||||
|
||||
# Read remaining output
|
||||
if self.process and self.process.stdout:
|
||||
if proc and proc.stdout:
|
||||
remaining = await loop.run_in_executor(
|
||||
None, self.process.stdout.read
|
||||
None, proc.stdout.read
|
||||
)
|
||||
if remaining:
|
||||
for line in remaining.strip().split('\n'):
|
||||
@@ -270,7 +358,7 @@ class CrawlerManager:
|
||||
|
||||
# Process ended
|
||||
if self.status == "running":
|
||||
exit_code = self.process.returncode if self.process else -1
|
||||
exit_code = proc.returncode if proc else -1
|
||||
if exit_code == 0:
|
||||
entry = self._create_log_entry("Crawler completed successfully", "success")
|
||||
else:
|
||||
@@ -283,6 +371,11 @@ class CrawlerManager:
|
||||
except Exception as e:
|
||||
entry = self._create_log_entry(f"Error reading output: {str(e)}", "error")
|
||||
await self._push_log(entry)
|
||||
finally:
|
||||
# Record the exit code and wake any run_and_wait() waiter. Runs in a
|
||||
# finally so a cancelled read task still releases the waiter.
|
||||
self.last_exit_code = proc.returncode if proc else None
|
||||
self._done.set()
|
||||
|
||||
|
||||
# Global singleton
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) 2025 [email protected]
|
||||
#
|
||||
# This file is part of MediaCrawler project.
|
||||
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/services/interpreter.py
|
||||
# GitHub: https://github.com/NanmiCoder
|
||||
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
|
||||
#
|
||||
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
|
||||
# 1. 不得用于任何商业用途。
|
||||
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
|
||||
# 3. 不得进行大规模爬取或对平台造成运营干扰。
|
||||
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
|
||||
# 5. 不得用于任何非法或不当的用途。
|
||||
#
|
||||
# 详细许可条款请参阅项目根目录下的LICENSE文件。
|
||||
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
|
||||
|
||||
"""Interpreter resolution for spawning crawler subprocesses.
|
||||
|
||||
Historically both the crawler manager and the environment check hardcoded
|
||||
``uv run``. ``uv`` is not guaranteed to be installed, so resolve the command
|
||||
prefix in one place: prefer ``uv`` (matching upstream docs), fall back to a
|
||||
project-local virtualenv, and finally to the interpreter running the server.
|
||||
"""
|
||||
|
||||
import shutil
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
# Project root: api/services/interpreter.py -> services -> api -> repo root
|
||||
PROJECT_ROOT = Path(__file__).parent.parent.parent
|
||||
|
||||
|
||||
def venv_python_path(project_root: Path | None = None) -> Path:
|
||||
"""Return the path to the project venv's Python executable."""
|
||||
root = project_root if project_root is not None else PROJECT_ROOT
|
||||
if sys.platform == "win32":
|
||||
return root / ".venv" / "Scripts" / "python.exe"
|
||||
return root / ".venv" / "bin" / "python"
|
||||
|
||||
|
||||
def resolve_python_cmd(project_root: Path | None = None) -> list[str]:
|
||||
"""Resolve the command prefix used to run ``main.py``.
|
||||
|
||||
Order of preference:
|
||||
1. ``uv`` if it is on PATH -- matches the upstream documented workflow.
|
||||
2. The project-local ``.venv`` if it exists.
|
||||
3. The interpreter currently running the API server.
|
||||
|
||||
Returns a list because the caller appends ``main.py`` and its flags.
|
||||
"""
|
||||
if shutil.which("uv"):
|
||||
return ["uv", "run", "python"]
|
||||
|
||||
venv_python = venv_python_path(project_root)
|
||||
if venv_python.exists():
|
||||
return [str(venv_python)]
|
||||
|
||||
return [sys.executable]
|
||||
|
||||
|
||||
def describe_interpreter(project_root: Path | None = None) -> str:
|
||||
"""Human-readable description of what resolve_python_cmd() picks."""
|
||||
cmd = resolve_python_cmd(project_root)
|
||||
if cmd[0] == "uv":
|
||||
return "uv run python"
|
||||
if cmd[0] == sys.executable:
|
||||
return f"current interpreter ({sys.executable})"
|
||||
return f"project virtualenv ({cmd[0]})"
|
||||
Reference in New Issue
Block a user