# -*- coding: utf-8 -*- # Copyright (c) 2025 relakkes@gmail.com # # This file is part of MediaCrawler project. # Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/covers.py # GitHub: https://github.com/NanmiCoder # Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 # # 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: # 1. 不得用于任何商业用途。 # 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 # 3. 不得进行大规模爬取或对平台造成运营干扰。 # 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 # 5. 不得用于任何非法或不当的用途。 # # 详细许可条款请参阅项目根目录下的LICENSE文件。 # 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 """作品封面本地缓存。 **为什么必须落盘**:小红书图床的地址是**带签名、会过期**的。路径里那段时间戳就是 签发时刻,实测: /202610080841/... (当天签发) → 200,且带不带 Referer 都 200 /202610070837/... (隔天) → 403,且带不带 Referer 都 403 所以这是**过期**,不是防盗链 —— 改 Referer 那一类修法治不了本。图一旦下载到本地, 就与签名无关,永远可读。 下载失败**不能影响采集**:一张封面拿不到,不该让整轮数据丢失。 """ import re from pathlib import Path from typing import Optional import httpx from .db import DATA_DIR COVERS_DIR = DATA_DIR / "covers" # 单张封面的上限。正常封面是几十 KB;超过这个数说明拿到的不是图, # 或者该放弃这一张而不是把内存撑爆。 MAX_COVER_BYTES = 5 * 1024 * 1024 USER_AGENT = ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/138.0.0.0 Safari/537.36" ) _EXTENSIONS = { "image/jpeg": ".jpg", "image/jpg": ".jpg", "image/png": ".png", "image/webp": ".webp", "image/gif": ".gif", "image/heic": ".heic", } # note_id 是平台的稳定标识,但仍要挡住路径穿越 —— 它会直接变成文件名。 _SAFE_ID = re.compile(r"^[A-Za-z0-9_-]{1,64}$") def is_safe_note_id(note_id: str) -> bool: return bool(note_id) and bool(_SAFE_ID.match(note_id)) def cache_dir() -> Path: COVERS_DIR.mkdir(parents=True, exist_ok=True) return COVERS_DIR def find_cached(note_id: str) -> Optional[Path]: """已缓存的封面文件,没有则 None。扩展名按内容类型而定,所以逐一试。""" if not is_safe_note_id(note_id): return None for extension in sorted(set(_EXTENSIONS.values())): candidate = COVERS_DIR / f"{note_id}{extension}" if candidate.is_file(): return candidate return None async def cache_cover(note_id: str, url: str) -> Optional[str]: """下载并保存一张封面,返回文件名;失败返回 None。 **从不抛异常**:调用方是采集入库流程,一张图拿不到不该让整轮数据出问题。 """ if not url or not is_safe_note_id(note_id): return None existing = find_cached(note_id) if existing is not None: return existing.name try: async with httpx.AsyncClient(timeout=20, follow_redirects=True) as client: response = await client.get(url, headers={"user-agent": USER_AGENT}) except Exception: return None if response.status_code != 200: # 403 通常意味着签名已过期 —— 这一张就没了,等下一轮采集拿到新地址。 return None content = response.content if not content or len(content) > MAX_COVER_BYTES: return None content_type = (response.headers.get("content-type") or "").split(";")[0].strip().lower() extension = _EXTENSIONS.get(content_type, ".jpg") # 图床偶尔不报 content-type,那种情况下扩展名只能猜,但文件本身仍然是好的。 target = cache_dir() / f"{note_id}{extension}" try: target.write_bytes(content) except OSError: return None return target.name def cover_url(note_id: str, remote: str) -> str: """前端该用哪个地址。 本地有缓存就用自己的接口 —— 那是唯一不会过期的地址。没有就退回远程地址, 至少让图先显示出来(哪怕它很快会失效)。 """ if find_cached(note_id) is not None: return f"/api/monitor/covers/{note_id}" return remote async def cache_pending(session, task_id: int, limit: int = 60) -> int: """把还没有本地副本的封面补下来,返回本次下载成功的张数。 由 runner 在入库之后调用,**而不是在 ingest 里** —— ingest 是刻意保持离线的 (它的文档写明 No network),往里塞网络请求会毁掉这一点。 每轮只补一批:一次跑几百张图既慢又会给图床压力,而旧地址本来就在陆续过期, 分摊到几轮里补完反而更稳。 """ from sqlalchemy import select from .models import MonitorNote notes = ( await session.scalars( select(MonitorNote) .where(MonitorNote.task_id == task_id, MonitorNote.cover != "") .order_by(MonitorNote.last_seen_at.desc()) .limit(limit) ) ).all() saved = 0 for note in notes: if find_cached(note.note_id) is not None: continue if await cache_cover(note.note_id, note.cover): saved += 1 return saved