在上游 MediaCrawler 之上新增一层: - 监控层 api/monitor/ —— 多博主/多笔记的定时采集、指标快照差分、报表、 企业微信通知。每轮采集写入独立目录,差分才成立。 - WebUI 登录鉴权 api/auth.py —— PBKDF2 口令 + 服务端会话,/api 全接口防护。 WebSocket 单独加依赖:BaseHTTPMiddleware 对 ws 作用域直接放行,覆盖不到。 - 全局平台切换 + 能力矩阵 —— 如实区分「爬虫模块支持」与「监控层已接线」, 未接通的平台直接拒绝建任务,而不是静默跑空。 - 监控库改用 MySQL 5.7(可回退 SQLite 供测试):逐表强制 utf8mb4 (服务端与库默认都是 latin1),启动校验所连 schema 以防写错库, 连接池 recycle + pre_ping 应对 MySQL 的 8 小时空闲断连。 修复上游缺陷: - xhs/core.py: 主页抓取失败会跳掉整个博主,导致一条作品都抓不到, 而那份资料只喂给一个空函数。改为尽力而为,失败不中断。 - xhs/login.py: cookie 登录只注入 web_session,冷启动签名会失败。 新增 INJECT_ALL_COOKIES 开关(默认关闭,原有行为不变)。 - requirements.txt: 补上 websockets。它在上游 pyproject.toml 里有声明、 这里漏了,导致 uvicorn 没有 WebSocket 能力,实时日志流从未工作。 改动过的上游文件清单及合并方式见 UPSTREAM.md。 测试:492 passed(另有 1 个既有的 Windows/gbk 上游测试失败,与本改动无关)
178 lines
5.9 KiB
Python
178 lines
5.9 KiB
Python
# -*- coding: utf-8 -*-
|
||
# Copyright (c) 2025 [email protected]
|
||
#
|
||
# This file is part of MediaCrawler project.
|
||
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/migrate_from_sqlite.py
|
||
# GitHub: https://github.com/NanmiCoder
|
||
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
|
||
#
|
||
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
|
||
# 1. 不得用于任何商业用途。
|
||
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
|
||
# 3. 不得进行大规模爬取或对平台造成运营干扰。
|
||
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
|
||
# 5. 不得用于任何非法或不当的用途。
|
||
#
|
||
# 详细许可条款请参阅项目根目录下的LICENSE文件。
|
||
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
|
||
|
||
"""One-off: copy the monitoring database from SQLite into MySQL.
|
||
|
||
python -m api.monitor.migrate_from_sqlite [--source data/monitor.db] [--dry-run]
|
||
|
||
Primary keys are preserved rather than reassigned, because rows in
|
||
``monitor_note`` / ``monitor_comment`` / ``monitor_run`` reference ``task_id``;
|
||
letting MySQL auto-assign new ids would silently break those links.
|
||
|
||
Refuses to run against a target that already holds data unless ``--force`` is
|
||
given, so a second accidental run cannot double everything up.
|
||
"""
|
||
|
||
import argparse
|
||
import sqlite3
|
||
import sys
|
||
from pathlib import Path
|
||
from typing import Any, Dict, List
|
||
|
||
PROJECT_ROOT = Path(__file__).parent.parent.parent
|
||
|
||
# Insert order matters: monitor_target and monitor_run carry real foreign keys to
|
||
# monitor_task, so the parent rows have to land first.
|
||
TABLES_IN_ORDER = [
|
||
"monitor_task",
|
||
"monitor_target",
|
||
"monitor_run",
|
||
"monitor_note",
|
||
"monitor_note_metric",
|
||
"monitor_comment",
|
||
"monitor_event",
|
||
"monitor_setting",
|
||
"auth_session",
|
||
]
|
||
|
||
|
||
def read_sqlite(path: Path) -> Dict[str, List[Dict[str, Any]]]:
|
||
if not path.exists():
|
||
raise SystemExit(f"找不到源库:{path}")
|
||
|
||
connection = sqlite3.connect(path)
|
||
connection.row_factory = sqlite3.Row
|
||
try:
|
||
existing = {
|
||
row[0]
|
||
for row in connection.execute(
|
||
"SELECT name FROM sqlite_master WHERE type='table'"
|
||
)
|
||
}
|
||
data: Dict[str, List[Dict[str, Any]]] = {}
|
||
for table in TABLES_IN_ORDER:
|
||
if table not in existing:
|
||
continue
|
||
rows = [dict(row) for row in connection.execute(f"SELECT * FROM {table}")]
|
||
if rows:
|
||
data[table] = rows
|
||
return data
|
||
finally:
|
||
connection.close()
|
||
|
||
|
||
def migrate(source: Path, dry_run: bool, force: bool) -> None:
|
||
import pymysql
|
||
|
||
from . import db as monitor_db
|
||
|
||
data = read_sqlite(source)
|
||
if not data:
|
||
print("源库里没有可迁移的数据。")
|
||
return
|
||
|
||
print("源库内容:")
|
||
for table, rows in data.items():
|
||
print(f" {table:22} {len(rows)} 行")
|
||
|
||
url = monitor_db.resolve_db_url()
|
||
if not url.startswith("mysql"):
|
||
raise SystemExit(f"目标不是 MySQL:{url}")
|
||
|
||
connection = pymysql.connect(
|
||
host=monitor_db.MYSQL_HOST(),
|
||
port=monitor_db.MYSQL_PORT(),
|
||
user=monitor_db.MYSQL_USER(),
|
||
password=monitor_db.MYSQL_PWD(),
|
||
database=monitor_db.MYSQL_DB_NAME(),
|
||
charset="utf8mb4",
|
||
autocommit=False,
|
||
)
|
||
|
||
try:
|
||
with connection.cursor() as cursor:
|
||
# Never write outside the configured schema.
|
||
cursor.execute("SELECT DATABASE()")
|
||
current = cursor.fetchone()[0]
|
||
expected = monitor_db.MYSQL_DB_NAME()
|
||
if current.lower() != expected.lower():
|
||
raise SystemExit(
|
||
f"当前连接的是 {current!r},配置要求 {expected!r};已中止。"
|
||
)
|
||
|
||
occupied = []
|
||
for table in data:
|
||
cursor.execute(f"SELECT COUNT(*) FROM `{table}`")
|
||
if cursor.fetchone()[0]:
|
||
occupied.append(table)
|
||
|
||
if occupied and not force:
|
||
raise SystemExit(
|
||
"目标库已有数据:" + ", ".join(occupied) + "\n"
|
||
"加 --force 才会继续(会与现有数据并存,造成重复)。"
|
||
)
|
||
|
||
if dry_run:
|
||
print("\n[试运行] 未写入任何数据。")
|
||
return
|
||
|
||
total = 0
|
||
for table, rows in data.items():
|
||
columns = list(rows[0].keys())
|
||
column_sql = ", ".join(f"`{c}`" for c in columns)
|
||
placeholders = ", ".join(["%s"] * len(columns))
|
||
statement = (
|
||
f"INSERT INTO `{table}` ({column_sql}) VALUES ({placeholders})"
|
||
)
|
||
cursor.executemany(
|
||
statement, [[row[c] for c in columns] for row in rows]
|
||
)
|
||
total += len(rows)
|
||
print(f" 已写入 {table:22} {len(rows)} 行")
|
||
|
||
connection.commit()
|
||
print(f"\n完成,共迁移 {total} 行。")
|
||
print("提示:源 SQLite 文件仍在原处,确认无误后自行删除。")
|
||
|
||
except Exception:
|
||
connection.rollback()
|
||
raise
|
||
finally:
|
||
connection.close()
|
||
|
||
|
||
def main(argv: List[str] | None = None) -> int:
|
||
parser = argparse.ArgumentParser(description="把监控库从 SQLite 迁到 MySQL")
|
||
parser.add_argument(
|
||
"--source",
|
||
default=str(PROJECT_ROOT / "data" / "monitor.db"),
|
||
help="SQLite 源文件路径",
|
||
)
|
||
parser.add_argument("--dry-run", action="store_true", help="只检查,不写入")
|
||
parser.add_argument(
|
||
"--force", action="store_true", help="目标库已有数据时也继续"
|
||
)
|
||
args = parser.parse_args(argv)
|
||
|
||
migrate(Path(args.source), args.dry_run, args.force)
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|