Files
MediaCrawler/api/monitor/migrate_from_sqlite.py
T
butubb 4e60524f37
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
feat: 监控面板 / 登录鉴权 / 多平台切换 / MySQL
在上游 MediaCrawler 之上新增一层:

- 监控层 api/monitor/ —— 多博主/多笔记的定时采集、指标快照差分、报表、
  企业微信通知。每轮采集写入独立目录,差分才成立。
- WebUI 登录鉴权 api/auth.py —— PBKDF2 口令 + 服务端会话,/api 全接口防护。
  WebSocket 单独加依赖:BaseHTTPMiddleware 对 ws 作用域直接放行,覆盖不到。
- 全局平台切换 + 能力矩阵 —— 如实区分「爬虫模块支持」与「监控层已接线」,
  未接通的平台直接拒绝建任务,而不是静默跑空。
- 监控库改用 MySQL 5.7(可回退 SQLite 供测试):逐表强制 utf8mb4
  (服务端与库默认都是 latin1),启动校验所连 schema 以防写错库,
  连接池 recycle + pre_ping 应对 MySQL 的 8 小时空闲断连。

修复上游缺陷:

- xhs/core.py: 主页抓取失败会跳掉整个博主,导致一条作品都抓不到,
  而那份资料只喂给一个空函数。改为尽力而为,失败不中断。
- xhs/login.py: cookie 登录只注入 web_session,冷启动签名会失败。
  新增 INJECT_ALL_COOKIES 开关(默认关闭,原有行为不变)。
- requirements.txt: 补上 websockets。它在上游 pyproject.toml 里有声明、
  这里漏了,导致 uvicorn 没有 WebSocket 能力,实时日志流从未工作。

改动过的上游文件清单及合并方式见 UPSTREAM.md。

测试:492 passed(另有 1 个既有的 Windows/gbk 上游测试失败,与本改动无关)
2026-10-07 09:58:40 +08:00

178 lines
5.9 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/migrate_from_sqlite.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""One-off: copy the monitoring database from SQLite into MySQL.
python -m api.monitor.migrate_from_sqlite [--source data/monitor.db] [--dry-run]
Primary keys are preserved rather than reassigned, because rows in
``monitor_note`` / ``monitor_comment`` / ``monitor_run`` reference ``task_id``;
letting MySQL auto-assign new ids would silently break those links.
Refuses to run against a target that already holds data unless ``--force`` is
given, so a second accidental run cannot double everything up.
"""
import argparse
import sqlite3
import sys
from pathlib import Path
from typing import Any, Dict, List
PROJECT_ROOT = Path(__file__).parent.parent.parent
# Insert order matters: monitor_target and monitor_run carry real foreign keys to
# monitor_task, so the parent rows have to land first.
TABLES_IN_ORDER = [
"monitor_task",
"monitor_target",
"monitor_run",
"monitor_note",
"monitor_note_metric",
"monitor_comment",
"monitor_event",
"monitor_setting",
"auth_session",
]
def read_sqlite(path: Path) -> Dict[str, List[Dict[str, Any]]]:
if not path.exists():
raise SystemExit(f"找不到源库:{path}")
connection = sqlite3.connect(path)
connection.row_factory = sqlite3.Row
try:
existing = {
row[0]
for row in connection.execute(
"SELECT name FROM sqlite_master WHERE type='table'"
)
}
data: Dict[str, List[Dict[str, Any]]] = {}
for table in TABLES_IN_ORDER:
if table not in existing:
continue
rows = [dict(row) for row in connection.execute(f"SELECT * FROM {table}")]
if rows:
data[table] = rows
return data
finally:
connection.close()
def migrate(source: Path, dry_run: bool, force: bool) -> None:
import pymysql
from . import db as monitor_db
data = read_sqlite(source)
if not data:
print("源库里没有可迁移的数据。")
return
print("源库内容:")
for table, rows in data.items():
print(f" {table:22} {len(rows)} 行")
url = monitor_db.resolve_db_url()
if not url.startswith("mysql"):
raise SystemExit(f"目标不是 MySQL:{url}")
connection = pymysql.connect(
host=monitor_db.MYSQL_HOST(),
port=monitor_db.MYSQL_PORT(),
user=monitor_db.MYSQL_USER(),
password=monitor_db.MYSQL_PWD(),
database=monitor_db.MYSQL_DB_NAME(),
charset="utf8mb4",
autocommit=False,
)
try:
with connection.cursor() as cursor:
# Never write outside the configured schema.
cursor.execute("SELECT DATABASE()")
current = cursor.fetchone()[0]
expected = monitor_db.MYSQL_DB_NAME()
if current.lower() != expected.lower():
raise SystemExit(
f"当前连接的是 {current!r},配置要求 {expected!r};已中止。"
)
occupied = []
for table in data:
cursor.execute(f"SELECT COUNT(*) FROM `{table}`")
if cursor.fetchone()[0]:
occupied.append(table)
if occupied and not force:
raise SystemExit(
"目标库已有数据:" + ", ".join(occupied) + "\n"
"加 --force 才会继续(会与现有数据并存,造成重复)。"
)
if dry_run:
print("\n[试运行] 未写入任何数据。")
return
total = 0
for table, rows in data.items():
columns = list(rows[0].keys())
column_sql = ", ".join(f"`{c}`" for c in columns)
placeholders = ", ".join(["%s"] * len(columns))
statement = (
f"INSERT INTO `{table}` ({column_sql}) VALUES ({placeholders})"
)
cursor.executemany(
statement, [[row[c] for c in columns] for row in rows]
)
total += len(rows)
print(f" 已写入 {table:22} {len(rows)} 行")
connection.commit()
print(f"\n完成,共迁移 {total} 行。")
print("提示:源 SQLite 文件仍在原处,确认无误后自行删除。")
except Exception:
connection.rollback()
raise
finally:
connection.close()
def main(argv: List[str] | None = None) -> int:
parser = argparse.ArgumentParser(description="把监控库从 SQLite 迁到 MySQL")
parser.add_argument(
"--source",
default=str(PROJECT_ROOT / "data" / "monitor.db"),
help="SQLite 源文件路径",
)
parser.add_argument("--dry-run", action="store_true", help="只检查,不写入")
parser.add_argument(
"--force", action="store_true", help="目标库已有数据时也继续"
)
args = parser.parse_args(argv)
migrate(Path(args.source), args.dry_run, args.force)
return 0
if __name__ == "__main__":
sys.exit(main())