diff --git a/api/monitor/covers.py b/api/monitor/covers.py
new file mode 100644
index 0000000..06b6fca
--- /dev/null
+++ b/api/monitor/covers.py
@@ -0,0 +1,162 @@
+# -*- coding: utf-8 -*-
+# Copyright (c) 2025 relakkes@gmail.com
+#
+# This file is part of MediaCrawler project.
+# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/covers.py
+# GitHub: https://github.com/NanmiCoder
+# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
+#
+# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
+# 1. 不得用于任何商业用途。
+# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
+# 3. 不得进行大规模爬取或对平台造成运营干扰。
+# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
+# 5. 不得用于任何非法或不当的用途。
+#
+# 详细许可条款请参阅项目根目录下的LICENSE文件。
+# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
+
+"""作品封面本地缓存。
+
+**为什么必须落盘**:小红书图床的地址是**带签名、会过期**的。路径里那段时间戳就是
+签发时刻,实测:
+
+ /202610080841/... (当天签发) → 200,且带不带 Referer 都 200
+ /202610070837/... (隔天) → 403,且带不带 Referer 都 403
+
+所以这是**过期**,不是防盗链 —— 改 Referer 那一类修法治不了本。图一旦下载到本地,
+就与签名无关,永远可读。
+
+下载失败**不能影响采集**:一张封面拿不到,不该让整轮数据丢失。
+"""
+
+import re
+from pathlib import Path
+from typing import Optional
+
+import httpx
+
+from .db import DATA_DIR
+
+COVERS_DIR = DATA_DIR / "covers"
+
+# 单张封面的上限。正常封面是几十 KB;超过这个数说明拿到的不是图,
+# 或者该放弃这一张而不是把内存撑爆。
+MAX_COVER_BYTES = 5 * 1024 * 1024
+
+USER_AGENT = (
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
+ "(KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36"
+)
+
+_EXTENSIONS = {
+ "image/jpeg": ".jpg",
+ "image/jpg": ".jpg",
+ "image/png": ".png",
+ "image/webp": ".webp",
+ "image/gif": ".gif",
+ "image/heic": ".heic",
+}
+
+# note_id 是平台的稳定标识,但仍要挡住路径穿越 —— 它会直接变成文件名。
+_SAFE_ID = re.compile(r"^[A-Za-z0-9_-]{1,64}$")
+
+
+def is_safe_note_id(note_id: str) -> bool:
+ return bool(note_id) and bool(_SAFE_ID.match(note_id))
+
+
+def cache_dir() -> Path:
+ COVERS_DIR.mkdir(parents=True, exist_ok=True)
+ return COVERS_DIR
+
+
+def find_cached(note_id: str) -> Optional[Path]:
+ """已缓存的封面文件,没有则 None。扩展名按内容类型而定,所以逐一试。"""
+ if not is_safe_note_id(note_id):
+ return None
+ for extension in sorted(set(_EXTENSIONS.values())):
+ candidate = COVERS_DIR / f"{note_id}{extension}"
+ if candidate.is_file():
+ return candidate
+ return None
+
+
+async def cache_cover(note_id: str, url: str) -> Optional[str]:
+ """下载并保存一张封面,返回文件名;失败返回 None。
+
+ **从不抛异常**:调用方是采集入库流程,一张图拿不到不该让整轮数据出问题。
+ """
+ if not url or not is_safe_note_id(note_id):
+ return None
+
+ existing = find_cached(note_id)
+ if existing is not None:
+ return existing.name
+
+ try:
+ async with httpx.AsyncClient(timeout=20, follow_redirects=True) as client:
+ response = await client.get(url, headers={"user-agent": USER_AGENT})
+ except Exception:
+ return None
+
+ if response.status_code != 200:
+ # 403 通常意味着签名已过期 —— 这一张就没了,等下一轮采集拿到新地址。
+ return None
+
+ content = response.content
+ if not content or len(content) > MAX_COVER_BYTES:
+ return None
+
+ content_type = (response.headers.get("content-type") or "").split(";")[0].strip().lower()
+ extension = _EXTENSIONS.get(content_type, ".jpg")
+ # 图床偶尔不报 content-type,那种情况下扩展名只能猜,但文件本身仍然是好的。
+ target = cache_dir() / f"{note_id}{extension}"
+
+ try:
+ target.write_bytes(content)
+ except OSError:
+ return None
+ return target.name
+
+
+def cover_url(note_id: str, remote: str) -> str:
+ """前端该用哪个地址。
+
+ 本地有缓存就用自己的接口 —— 那是唯一不会过期的地址。没有就退回远程地址,
+ 至少让图先显示出来(哪怕它很快会失效)。
+ """
+ if find_cached(note_id) is not None:
+ return f"/api/monitor/covers/{note_id}"
+ return remote
+
+
+async def cache_pending(session, task_id: int, limit: int = 60) -> int:
+ """把还没有本地副本的封面补下来,返回本次下载成功的张数。
+
+ 由 runner 在入库之后调用,**而不是在 ingest 里** —— ingest 是刻意保持离线的
+ (它的文档写明 No network),往里塞网络请求会毁掉这一点。
+
+ 每轮只补一批:一次跑几百张图既慢又会给图床压力,而旧地址本来就在陆续过期,
+ 分摊到几轮里补完反而更稳。
+ """
+ from sqlalchemy import select
+
+ from .models import MonitorNote
+
+ notes = (
+ await session.scalars(
+ select(MonitorNote)
+ .where(MonitorNote.task_id == task_id, MonitorNote.cover != "")
+ .order_by(MonitorNote.last_seen_at.desc())
+ .limit(limit)
+ )
+ ).all()
+
+ saved = 0
+ for note in notes:
+ if find_cached(note.note_id) is not None:
+ continue
+ if await cache_cover(note.note_id, note.cover):
+ saved += 1
+ return saved
diff --git a/api/monitor/ingest.py b/api/monitor/ingest.py
index a49d797..fe30907 100644
--- a/api/monitor/ingest.py
+++ b/api/monitor/ingest.py
@@ -323,6 +323,11 @@ async def _ingest_notes(
# Only refresh descriptive fields; seen-tracking is updated below.
if title:
note.title = title
+ # 封面地址**带签名、会过期**,所以每轮都用最新的覆盖它。原先只在首次入库
+ # 时写一次,结果旧作品的封面地址烂在库里 —— 隔天开始全是 403,而且再怎么
+ # 重跑也修不回来。落盘那份由 covers.cache_pending 负责(网络操作不在本模块)。
+ if cover:
+ note.cover = cover
note.last_seen_run_id = run.id
note.last_seen_at = now
diff --git a/api/monitor/runner.py b/api/monitor/runner.py
index 9ed077e..5a55d6f 100644
--- a/api/monitor/runner.py
+++ b/api/monitor/runner.py
@@ -38,7 +38,7 @@ from ..schemas import (
SaveDataOptionEnum,
)
from ..services import crawler_manager
-from . import app_settings, notify
+from . import app_settings, covers, notify
from .db import get_session
from .ingest import IngestResult, ingest_run
from .models import (
@@ -280,4 +280,15 @@ async def execute_task(task_id: int, trigger: str = "manual") -> IngestResult:
if task is not None and run is not None:
await notify.notify_run(session, task, run)
+ # --- Phase 5: 封面落盘 ------------------------------------------------------
+ # 也放在事务之外。封面地址带签名、会过期(实测隔天即 403),落盘之后才与签名无关。
+ # 下载慢且可能失败,占着一个入库事务是不合适的;失败也不影响本轮数据。
+ try:
+ async with get_session() as session:
+ saved = await covers.cache_pending(session, task_id)
+ if saved:
+ print(f"[monitor.runner] 缓存了 {saved} 张作品封面")
+ except Exception as exc: # noqa: BLE001 - 封面拿不到不该让整轮失败
+ print(f"[monitor.runner] 封面缓存失败:{exc}")
+
return result
diff --git a/api/monitor/service.py b/api/monitor/service.py
index 117e81d..74d34b6 100644
--- a/api/monitor/service.py
+++ b/api/monitor/service.py
@@ -28,7 +28,7 @@ from sqlalchemy.ext.asyncio import AsyncSession
from tools.time_util import get_current_timestamp
-from . import app_settings, platforms, schedule
+from . import app_settings, covers, platforms, schedule
from .db import get_session
from .platforms import PLATFORM_XHS
from .models import (
@@ -397,7 +397,9 @@ async def list_notes(
"note_id": note.note_id,
"title": note.title,
"note_url": note.note_url,
- "cover": note.cover,
+ # 优先给本地缓存地址:远程地址带签名、会过期(实测隔天即 403),
+ # 本地那份不会。没有缓存时才退回远程,至少让图先显示出来。
+ "cover": covers.cover_url(note.note_id, note.cover),
"first_seen_at": note.first_seen_at,
"last_seen_at": note.last_seen_at,
"is_new": note.first_seen_run_id == latest_run_ids.get(note.task_id),
@@ -473,7 +475,7 @@ async def _note_meta_map(
return {
row.note_id: {
"note_title": row.title,
- "note_cover": row.cover,
+ "note_cover": covers.cover_url(row.note_id, row.cover),
"note_url": row.note_url,
"task_id": row.task_id,
}
diff --git a/api/routers/monitor.py b/api/routers/monitor.py
index 79592c8..656bcf4 100644
--- a/api/routers/monitor.py
+++ b/api/routers/monitor.py
@@ -22,8 +22,9 @@ from datetime import date, timedelta
from typing import Any, Dict, List, Optional
from fastapi import APIRouter, HTTPException, Query, Response
+from fastapi.responses import FileResponse
-from ..monitor import notify, qrlogin, report, service
+from ..monitor import covers, notify, qrlogin, report, service
from ..monitor.db import get_session
from ..monitor.platforms import PLATFORM_XHS
from ..monitor.settings import (
@@ -252,6 +253,35 @@ async def clear_cookie_endpoint(platform: str = Query(default=PLATFORM_XHS)):
# Scanning once is what makes later unattended runs logged in.
+@router.get("/covers/{note_id}")
+async def get_cover(note_id: str):
+ """作品封面,从本地缓存读。
+
+ **为什么不让前端直连图床**:图床地址是带签名、会过期的 —— 实测隔天即 403,
+ 而且带不带 Referer 都一样,所以那是过期而不是防盗链。本地那份与签名无关。
+
+ 这个路由是带鉴权的(整条 monitor 路由都挂了 require_auth),所以封面不会被
+ 匿名读走;前端用同源的
请求会自动带上会话 cookie。
+ """
+ path = covers.find_cached(note_id)
+ if path is None:
+ raise HTTPException(status_code=404, detail="封面未缓存")
+
+ media_types = {
+ ".jpg": "image/jpeg",
+ ".png": "image/png",
+ ".webp": "image/webp",
+ ".gif": "image/gif",
+ ".heic": "image/heic",
+ }
+ return FileResponse(
+ path,
+ media_type=media_types.get(path.suffix.lower(), "application/octet-stream"),
+ # 本地文件不会变(note_id 唯一),让浏览器自己缓存,省掉重复请求。
+ headers={"Cache-Control": "private, max-age=86400"},
+ )
+
+
@router.post("/login/qr")
async def start_qr_login(platform: str = Query(default=PLATFORM_XHS)):
"""Open the login page in the CDP browser and return its QR code.
diff --git a/webui/src/components/monitor/NoteCover.tsx b/webui/src/components/monitor/NoteCover.tsx
index 7968292..8222286 100644
--- a/webui/src/components/monitor/NoteCover.tsx
+++ b/webui/src/components/monitor/NoteCover.tsx
@@ -1,14 +1,17 @@
/**
* 作品封面缩略图。
*
- * `referrerPolicy="no-referrer"` 是功能性的,不是装饰。小红书的图床对带**外部
- * Referer** 的请求一律返回 403,而浏览器对跨域 `
` 默认就会带上本站的源作为
- * Referer —— 于是封面全都显示成破图,而 URL 本身完全正常。实测同一张图:
+ * **图为什么曾经全是破图**:不是防盗链。小红书图床的地址**带签名、会过期** ——
+ * 路径里那段时间戳就是签发时刻。实测同一批图:
*
- * 无 Referer -> 200, 67968 B, image/webp
- * Referer 为本站 -> 403, 0 B
+ * 当天签发的地址 -> 200,带不带 Referer 都一样
+ * 隔天的地址 -> 403,带不带 Referer 都一样
*
- * 放在一个组件里,是为了让下一个要显示封面的页面不会漏掉这个属性。
+ * 所以 Referer 根本不是那个维度,改它是白改。真正的解法是后端把图**下载到本地**,
+ * 前端拿到的是 `/api/monitor/covers/{note_id}` —— 与签名无关,不会过期。
+ *
+ * `referrerPolicy` 保留着,是因为它仍然是个合理的默认(外链图不该把自己的地址
+ * 泄露给第三方),但**它不再是这里能正常显示的原因**。
*/
const SIZES = {