fix(covers): 封面地址是签名过期而非防盗链,改为本地缓存
实测推翻了之前的诊断。同一批图:
当天签发的地址 /202610080841/... -> 200,带不带 Referer 都一样
隔天的地址 /202610070837/... -> 403,带不带 Referer 都一样
路径里那段时间戳就是签发时刻。所以这是**过期**,Referer 根本不是那个维度 ——
上一轮加 referrerPolicy 是照着错误结论改的,白改。
修法:
- 采集入库时每轮刷新 cover 地址。原先只在首次入库写一次,旧作品的地址烂在库里,
而且再怎么重跑也修不回来
- 新增 api/monitor/covers.py:把图下载落盘。图一旦落盘就与签名无关,永远可读
- 下载放在 runner 的 Phase 5(事务已提交之后),不放 ingest —— ingest 的文档写明
No network,往里塞网络请求会毁掉它可离线测试这一点
- 新增 GET /api/monitor/covers/{note_id} 取图。这条路由带鉴权,封面不会被匿名读走
- service 返回本地地址优先,没有缓存时才退回远程
- 每轮只补一批(60 张):一次跑几百张既慢又会给图床压力,而旧地址本来就在陆续过期,
分摊到几轮反而更稳
顺带修正 NoteCover 的注释 —— 它写着防盗链,而那个结论已被推翻,留个错的注释比没有更糟。
This commit is contained in:
@@ -0,0 +1,162 @@
|
|||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
# Copyright (c) 2025 [email protected]
|
||||||
|
#
|
||||||
|
# This file is part of MediaCrawler project.
|
||||||
|
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/covers.py
|
||||||
|
# GitHub: https://github.com/NanmiCoder
|
||||||
|
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
|
||||||
|
#
|
||||||
|
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
|
||||||
|
# 1. 不得用于任何商业用途。
|
||||||
|
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
|
||||||
|
# 3. 不得进行大规模爬取或对平台造成运营干扰。
|
||||||
|
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
|
||||||
|
# 5. 不得用于任何非法或不当的用途。
|
||||||
|
#
|
||||||
|
# 详细许可条款请参阅项目根目录下的LICENSE文件。
|
||||||
|
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
|
||||||
|
|
||||||
|
"""作品封面本地缓存。
|
||||||
|
|
||||||
|
**为什么必须落盘**:小红书图床的地址是**带签名、会过期**的。路径里那段时间戳就是
|
||||||
|
签发时刻,实测:
|
||||||
|
|
||||||
|
/202610080841/... (当天签发) → 200,且带不带 Referer 都 200
|
||||||
|
/202610070837/... (隔天) → 403,且带不带 Referer 都 403
|
||||||
|
|
||||||
|
所以这是**过期**,不是防盗链 —— 改 Referer 那一类修法治不了本。图一旦下载到本地,
|
||||||
|
就与签名无关,永远可读。
|
||||||
|
|
||||||
|
下载失败**不能影响采集**:一张封面拿不到,不该让整轮数据丢失。
|
||||||
|
"""
|
||||||
|
|
||||||
|
import re
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Optional
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from .db import DATA_DIR
|
||||||
|
|
||||||
|
COVERS_DIR = DATA_DIR / "covers"
|
||||||
|
|
||||||
|
# 单张封面的上限。正常封面是几十 KB;超过这个数说明拿到的不是图,
|
||||||
|
# 或者该放弃这一张而不是把内存撑爆。
|
||||||
|
MAX_COVER_BYTES = 5 * 1024 * 1024
|
||||||
|
|
||||||
|
USER_AGENT = (
|
||||||
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||||
|
"(KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
_EXTENSIONS = {
|
||||||
|
"image/jpeg": ".jpg",
|
||||||
|
"image/jpg": ".jpg",
|
||||||
|
"image/png": ".png",
|
||||||
|
"image/webp": ".webp",
|
||||||
|
"image/gif": ".gif",
|
||||||
|
"image/heic": ".heic",
|
||||||
|
}
|
||||||
|
|
||||||
|
# note_id 是平台的稳定标识,但仍要挡住路径穿越 —— 它会直接变成文件名。
|
||||||
|
_SAFE_ID = re.compile(r"^[A-Za-z0-9_-]{1,64}$")
|
||||||
|
|
||||||
|
|
||||||
|
def is_safe_note_id(note_id: str) -> bool:
|
||||||
|
return bool(note_id) and bool(_SAFE_ID.match(note_id))
|
||||||
|
|
||||||
|
|
||||||
|
def cache_dir() -> Path:
|
||||||
|
COVERS_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
return COVERS_DIR
|
||||||
|
|
||||||
|
|
||||||
|
def find_cached(note_id: str) -> Optional[Path]:
|
||||||
|
"""已缓存的封面文件,没有则 None。扩展名按内容类型而定,所以逐一试。"""
|
||||||
|
if not is_safe_note_id(note_id):
|
||||||
|
return None
|
||||||
|
for extension in sorted(set(_EXTENSIONS.values())):
|
||||||
|
candidate = COVERS_DIR / f"{note_id}{extension}"
|
||||||
|
if candidate.is_file():
|
||||||
|
return candidate
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
async def cache_cover(note_id: str, url: str) -> Optional[str]:
|
||||||
|
"""下载并保存一张封面,返回文件名;失败返回 None。
|
||||||
|
|
||||||
|
**从不抛异常**:调用方是采集入库流程,一张图拿不到不该让整轮数据出问题。
|
||||||
|
"""
|
||||||
|
if not url or not is_safe_note_id(note_id):
|
||||||
|
return None
|
||||||
|
|
||||||
|
existing = find_cached(note_id)
|
||||||
|
if existing is not None:
|
||||||
|
return existing.name
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with httpx.AsyncClient(timeout=20, follow_redirects=True) as client:
|
||||||
|
response = await client.get(url, headers={"user-agent": USER_AGENT})
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
|
||||||
|
if response.status_code != 200:
|
||||||
|
# 403 通常意味着签名已过期 —— 这一张就没了,等下一轮采集拿到新地址。
|
||||||
|
return None
|
||||||
|
|
||||||
|
content = response.content
|
||||||
|
if not content or len(content) > MAX_COVER_BYTES:
|
||||||
|
return None
|
||||||
|
|
||||||
|
content_type = (response.headers.get("content-type") or "").split(";")[0].strip().lower()
|
||||||
|
extension = _EXTENSIONS.get(content_type, ".jpg")
|
||||||
|
# 图床偶尔不报 content-type,那种情况下扩展名只能猜,但文件本身仍然是好的。
|
||||||
|
target = cache_dir() / f"{note_id}{extension}"
|
||||||
|
|
||||||
|
try:
|
||||||
|
target.write_bytes(content)
|
||||||
|
except OSError:
|
||||||
|
return None
|
||||||
|
return target.name
|
||||||
|
|
||||||
|
|
||||||
|
def cover_url(note_id: str, remote: str) -> str:
|
||||||
|
"""前端该用哪个地址。
|
||||||
|
|
||||||
|
本地有缓存就用自己的接口 —— 那是唯一不会过期的地址。没有就退回远程地址,
|
||||||
|
至少让图先显示出来(哪怕它很快会失效)。
|
||||||
|
"""
|
||||||
|
if find_cached(note_id) is not None:
|
||||||
|
return f"/api/monitor/covers/{note_id}"
|
||||||
|
return remote
|
||||||
|
|
||||||
|
|
||||||
|
async def cache_pending(session, task_id: int, limit: int = 60) -> int:
|
||||||
|
"""把还没有本地副本的封面补下来,返回本次下载成功的张数。
|
||||||
|
|
||||||
|
由 runner 在入库之后调用,**而不是在 ingest 里** —— ingest 是刻意保持离线的
|
||||||
|
(它的文档写明 No network),往里塞网络请求会毁掉这一点。
|
||||||
|
|
||||||
|
每轮只补一批:一次跑几百张图既慢又会给图床压力,而旧地址本来就在陆续过期,
|
||||||
|
分摊到几轮里补完反而更稳。
|
||||||
|
"""
|
||||||
|
from sqlalchemy import select
|
||||||
|
|
||||||
|
from .models import MonitorNote
|
||||||
|
|
||||||
|
notes = (
|
||||||
|
await session.scalars(
|
||||||
|
select(MonitorNote)
|
||||||
|
.where(MonitorNote.task_id == task_id, MonitorNote.cover != "")
|
||||||
|
.order_by(MonitorNote.last_seen_at.desc())
|
||||||
|
.limit(limit)
|
||||||
|
)
|
||||||
|
).all()
|
||||||
|
|
||||||
|
saved = 0
|
||||||
|
for note in notes:
|
||||||
|
if find_cached(note.note_id) is not None:
|
||||||
|
continue
|
||||||
|
if await cache_cover(note.note_id, note.cover):
|
||||||
|
saved += 1
|
||||||
|
return saved
|
||||||
@@ -323,6 +323,11 @@ async def _ingest_notes(
|
|||||||
# Only refresh descriptive fields; seen-tracking is updated below.
|
# Only refresh descriptive fields; seen-tracking is updated below.
|
||||||
if title:
|
if title:
|
||||||
note.title = title
|
note.title = title
|
||||||
|
# 封面地址**带签名、会过期**,所以每轮都用最新的覆盖它。原先只在首次入库
|
||||||
|
# 时写一次,结果旧作品的封面地址烂在库里 —— 隔天开始全是 403,而且再怎么
|
||||||
|
# 重跑也修不回来。落盘那份由 covers.cache_pending 负责(网络操作不在本模块)。
|
||||||
|
if cover:
|
||||||
|
note.cover = cover
|
||||||
note.last_seen_run_id = run.id
|
note.last_seen_run_id = run.id
|
||||||
note.last_seen_at = now
|
note.last_seen_at = now
|
||||||
|
|
||||||
|
|||||||
+12
-1
@@ -38,7 +38,7 @@ from ..schemas import (
|
|||||||
SaveDataOptionEnum,
|
SaveDataOptionEnum,
|
||||||
)
|
)
|
||||||
from ..services import crawler_manager
|
from ..services import crawler_manager
|
||||||
from . import app_settings, notify
|
from . import app_settings, covers, notify
|
||||||
from .db import get_session
|
from .db import get_session
|
||||||
from .ingest import IngestResult, ingest_run
|
from .ingest import IngestResult, ingest_run
|
||||||
from .models import (
|
from .models import (
|
||||||
@@ -280,4 +280,15 @@ async def execute_task(task_id: int, trigger: str = "manual") -> IngestResult:
|
|||||||
if task is not None and run is not None:
|
if task is not None and run is not None:
|
||||||
await notify.notify_run(session, task, run)
|
await notify.notify_run(session, task, run)
|
||||||
|
|
||||||
|
# --- Phase 5: 封面落盘 ------------------------------------------------------
|
||||||
|
# 也放在事务之外。封面地址带签名、会过期(实测隔天即 403),落盘之后才与签名无关。
|
||||||
|
# 下载慢且可能失败,占着一个入库事务是不合适的;失败也不影响本轮数据。
|
||||||
|
try:
|
||||||
|
async with get_session() as session:
|
||||||
|
saved = await covers.cache_pending(session, task_id)
|
||||||
|
if saved:
|
||||||
|
print(f"[monitor.runner] 缓存了 {saved} 张作品封面")
|
||||||
|
except Exception as exc: # noqa: BLE001 - 封面拿不到不该让整轮失败
|
||||||
|
print(f"[monitor.runner] 封面缓存失败:{exc}")
|
||||||
|
|
||||||
return result
|
return result
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ from sqlalchemy.ext.asyncio import AsyncSession
|
|||||||
|
|
||||||
from tools.time_util import get_current_timestamp
|
from tools.time_util import get_current_timestamp
|
||||||
|
|
||||||
from . import app_settings, platforms, schedule
|
from . import app_settings, covers, platforms, schedule
|
||||||
from .db import get_session
|
from .db import get_session
|
||||||
from .platforms import PLATFORM_XHS
|
from .platforms import PLATFORM_XHS
|
||||||
from .models import (
|
from .models import (
|
||||||
@@ -397,7 +397,9 @@ async def list_notes(
|
|||||||
"note_id": note.note_id,
|
"note_id": note.note_id,
|
||||||
"title": note.title,
|
"title": note.title,
|
||||||
"note_url": note.note_url,
|
"note_url": note.note_url,
|
||||||
"cover": note.cover,
|
# 优先给本地缓存地址:远程地址带签名、会过期(实测隔天即 403),
|
||||||
|
# 本地那份不会。没有缓存时才退回远程,至少让图先显示出来。
|
||||||
|
"cover": covers.cover_url(note.note_id, note.cover),
|
||||||
"first_seen_at": note.first_seen_at,
|
"first_seen_at": note.first_seen_at,
|
||||||
"last_seen_at": note.last_seen_at,
|
"last_seen_at": note.last_seen_at,
|
||||||
"is_new": note.first_seen_run_id == latest_run_ids.get(note.task_id),
|
"is_new": note.first_seen_run_id == latest_run_ids.get(note.task_id),
|
||||||
@@ -473,7 +475,7 @@ async def _note_meta_map(
|
|||||||
return {
|
return {
|
||||||
row.note_id: {
|
row.note_id: {
|
||||||
"note_title": row.title,
|
"note_title": row.title,
|
||||||
"note_cover": row.cover,
|
"note_cover": covers.cover_url(row.note_id, row.cover),
|
||||||
"note_url": row.note_url,
|
"note_url": row.note_url,
|
||||||
"task_id": row.task_id,
|
"task_id": row.task_id,
|
||||||
}
|
}
|
||||||
|
|||||||
+31
-1
@@ -22,8 +22,9 @@ from datetime import date, timedelta
|
|||||||
from typing import Any, Dict, List, Optional
|
from typing import Any, Dict, List, Optional
|
||||||
|
|
||||||
from fastapi import APIRouter, HTTPException, Query, Response
|
from fastapi import APIRouter, HTTPException, Query, Response
|
||||||
|
from fastapi.responses import FileResponse
|
||||||
|
|
||||||
from ..monitor import notify, qrlogin, report, service
|
from ..monitor import covers, notify, qrlogin, report, service
|
||||||
from ..monitor.db import get_session
|
from ..monitor.db import get_session
|
||||||
from ..monitor.platforms import PLATFORM_XHS
|
from ..monitor.platforms import PLATFORM_XHS
|
||||||
from ..monitor.settings import (
|
from ..monitor.settings import (
|
||||||
@@ -252,6 +253,35 @@ async def clear_cookie_endpoint(platform: str = Query(default=PLATFORM_XHS)):
|
|||||||
# Scanning once is what makes later unattended runs logged in.
|
# Scanning once is what makes later unattended runs logged in.
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/covers/{note_id}")
|
||||||
|
async def get_cover(note_id: str):
|
||||||
|
"""作品封面,从本地缓存读。
|
||||||
|
|
||||||
|
**为什么不让前端直连图床**:图床地址是带签名、会过期的 —— 实测隔天即 403,
|
||||||
|
而且带不带 Referer 都一样,所以那是过期而不是防盗链。本地那份与签名无关。
|
||||||
|
|
||||||
|
这个路由是带鉴权的(整条 monitor 路由都挂了 require_auth),所以封面不会被
|
||||||
|
匿名读走;前端用同源的 <img> 请求会自动带上会话 cookie。
|
||||||
|
"""
|
||||||
|
path = covers.find_cached(note_id)
|
||||||
|
if path is None:
|
||||||
|
raise HTTPException(status_code=404, detail="封面未缓存")
|
||||||
|
|
||||||
|
media_types = {
|
||||||
|
".jpg": "image/jpeg",
|
||||||
|
".png": "image/png",
|
||||||
|
".webp": "image/webp",
|
||||||
|
".gif": "image/gif",
|
||||||
|
".heic": "image/heic",
|
||||||
|
}
|
||||||
|
return FileResponse(
|
||||||
|
path,
|
||||||
|
media_type=media_types.get(path.suffix.lower(), "application/octet-stream"),
|
||||||
|
# 本地文件不会变(note_id 唯一),让浏览器自己缓存,省掉重复请求。
|
||||||
|
headers={"Cache-Control": "private, max-age=86400"},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@router.post("/login/qr")
|
@router.post("/login/qr")
|
||||||
async def start_qr_login(platform: str = Query(default=PLATFORM_XHS)):
|
async def start_qr_login(platform: str = Query(default=PLATFORM_XHS)):
|
||||||
"""Open the login page in the CDP browser and return its QR code.
|
"""Open the login page in the CDP browser and return its QR code.
|
||||||
|
|||||||
@@ -1,14 +1,17 @@
|
|||||||
/**
|
/**
|
||||||
* 作品封面缩略图。
|
* 作品封面缩略图。
|
||||||
*
|
*
|
||||||
* `referrerPolicy="no-referrer"` 是功能性的,不是装饰。小红书的图床对带**外部
|
* **图为什么曾经全是破图**:不是防盗链。小红书图床的地址**带签名、会过期** ——
|
||||||
* Referer** 的请求一律返回 403,而浏览器对跨域 `<img>` 默认就会带上本站的源作为
|
* 路径里那段时间戳就是签发时刻。实测同一批图:
|
||||||
* Referer —— 于是封面全都显示成破图,而 URL 本身完全正常。实测同一张图:
|
|
||||||
*
|
*
|
||||||
* 无 Referer -> 200, 67968 B, image/webp
|
* 当天签发的地址 -> 200,带不带 Referer 都一样
|
||||||
* Referer 为本站 -> 403, 0 B
|
* 隔天的地址 -> 403,带不带 Referer 都一样
|
||||||
*
|
*
|
||||||
* 放在一个组件里,是为了让下一个要显示封面的页面不会漏掉这个属性。
|
* 所以 Referer 根本不是那个维度,改它是白改。真正的解法是后端把图**下载到本地**,
|
||||||
|
* 前端拿到的是 `/api/monitor/covers/{note_id}` —— 与签名无关,不会过期。
|
||||||
|
*
|
||||||
|
* `referrerPolicy` 保留着,是因为它仍然是个合理的默认(外链图不该把自己的地址
|
||||||
|
* 泄露给第三方),但**它不再是这里能正常显示的原因**。
|
||||||
*/
|
*/
|
||||||
|
|
||||||
const SIZES = {
|
const SIZES = {
|
||||||
|
|||||||
Reference in New Issue
Block a user