Files
MediaCrawler/api/creator/models.py
T
butubb fa227600fe
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s
fix(creator): 权限开通后的状态显示 + 同步范围可选可查
- 权限显示:实测今天 display/status 已变成 1,但 tip 是「数据正在更新中,请耐心等待」,
  作品仍是 0 条。原逻辑只在未开通时显示提示条,于是会一边写着「数据已开通」一边列不出
  作品,自相矛盾。现在只要后台有话要说就显示,并按状态区分措辞。
- 同步范围:原本写死 90 天,界面上既看不见也改不了。现在可选 7/30/90/180/365/730 天,
  与后端校验上限一致。
- 新增 creator_account.last_sync_days,记录上次实际使用的范围 —— 否则界面只能说
  「同步过了」,说不清覆盖的是哪一段。选择器也会对齐到它,避免上次同步 1 年、这次
  点一下悄悄缩回 90 天。
2026-10-08 07:52:28 +08:00

133 lines
6.5 KiB
Python

# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/creator/models.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""运营模块的数据模型。
**刻意复用 `MonitorBase`**:这样 `init_db` 的 `create_all` 会顺手建出新表,而
`_ensure_columns`(已改为按模型元数据推导)也会自动给新表补字段 —— 不必再维护一份
建表语句。表落在同一个库里,与监控互不干扰。
"""
from typing import Optional
from sqlalchemy import BigInteger, Float, ForeignKey, Index, Integer, String, Text
from sqlalchemy.orm import Mapped, mapped_column, relationship
from ..monitor.models import MonitorBase
# 创作者后台的数据权限是「首次访问时自动申请、次日生效」。这个状态必须如实呈现:
# 显示成"没数据"会让人以为采集坏了,实际是在等审批。
PERMISSION_UNKNOWN = "unknown"
PERMISSION_PENDING = "pending" # 已申请,未生效(提示语:"次日可查看")
PERMISSION_ACTIVE = "active"
PERMISSION_MISSING = "missing" # 接口明确说没有权限
# 账号自身的可用性。
ACCOUNT_OK = "ok"
ACCOUNT_EXPIRED = "expired" # cookie 失效,需要重新扫码
ACCOUNT_ERROR = "error"
class CreatorAccount(MonitorBase):
"""一个自己的小红书账号。
纯请求路线下,**一个账号的全部身份就是一份 cookie** —— 没有浏览器 profile、
没有独立目录。所以"多账号"在这里只是表里的多行,不是多套运行环境。
`cookie` 是凭证:与监控的 cookie 同样对待,只存库、绝不回显接口。
"""
__tablename__ = "creator_account"
id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True)
# 展示名。优先用后台返回的昵称,用户可以改。
nickname: Mapped[str] = mapped_column(String(128), nullable=False, default="")
# 创作者后台的账号标识,由 /api/galaxy/user/info 返回,用于去重。
user_id: Mapped[str] = mapped_column(String(64), nullable=False, default="", index=True)
red_id: Mapped[str] = mapped_column(String(64), nullable=False, default="")
avatar: Mapped[str] = mapped_column(Text, nullable=False, default="")
cookie: Mapped[str] = mapped_column(Text, nullable=False, default="")
status: Mapped[str] = mapped_column(String(16), nullable=False, default=ACCOUNT_OK)
permission_status: Mapped[str] = mapped_column(
String(16), nullable=False, default=PERMISSION_UNKNOWN
)
# 后台原话,例如"已为您申请数据权限,次日可查看"。照抄,不改写。
permission_tip: Mapped[str] = mapped_column(Text, nullable=False, default="")
last_checked_at: Mapped[Optional[int]] = mapped_column(BigInteger)
last_synced_at: Mapped[Optional[int]] = mapped_column(BigInteger)
# 上次同步用的时间范围(天,按发布时间)。当前展示的数据就是这个范围的产物 ——
# 不记下来的话,界面只能说明"同步过了",说不清是哪一段。
last_sync_days: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
last_error: Mapped[Optional[str]] = mapped_column(Text)
created_at: Mapped[int] = mapped_column(BigInteger, nullable=False)
updated_at: Mapped[int] = mapped_column(BigInteger, nullable=False)
notes: Mapped[list["CreatorNoteStat"]] = relationship(
back_populates="account", cascade="all, delete-orphan"
)
class CreatorNoteStat(MonitorBase):
"""一篇作品在某个采集时点的运营数据。
创作者后台给的是**累计值**(截至查询时点),所以反复采集天然形成时间序列 ——
与监控的"快照 + 差分"是同一个思路,因此这里保留 `captured_at` 而不是覆盖写。
"""
__tablename__ = "creator_note_stat"
id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True)
account_id: Mapped[int] = mapped_column(
ForeignKey("creator_account.id", ondelete="CASCADE"), nullable=False, index=True
)
note_id: Mapped[str] = mapped_column(String(64), nullable=False, index=True)
title: Mapped[str] = mapped_column(Text, nullable=False, default="")
# 发布时间(毫秒)。后台按发布时间筛选,这是它的主时间轴。
publish_time: Mapped[Optional[int]] = mapped_column(BigInteger)
# --- 运营指标 ---------------------------------------------------------
# 计数用 BigInteger:曝光量可以很大,用 INT 迟早溢出。
exposure: Mapped[Optional[int]] = mapped_column(BigInteger)
views: Mapped[Optional[int]] = mapped_column(BigInteger)
likes: Mapped[Optional[int]] = mapped_column(BigInteger)
comments: Mapped[Optional[int]] = mapped_column(BigInteger)
favorites: Mapped[Optional[int]] = mapped_column(BigInteger)
shares: Mapped[Optional[int]] = mapped_column(BigInteger)
new_followers: Mapped[Optional[int]] = mapped_column(BigInteger)
danmaku: Mapped[Optional[int]] = mapped_column(BigInteger)
# 比率与时长。后台返回的可能是 "12.3%"/"1分30秒" 这类字符串,解析不了的存 NULL
# 而不是 0 —— 与监控层的口径一致:0 是真实值,NULL 是"不知道"。
cover_ctr: Mapped[Optional[float]] = mapped_column(Float)
avg_watch_seconds: Mapped[Optional[float]] = mapped_column(Float)
two_second_exit_rate: Mapped[Optional[float]] = mapped_column(Float)
completion_rate: Mapped[Optional[float]] = mapped_column(Float)
captured_at: Mapped[int] = mapped_column(BigInteger, nullable=False, index=True)
account: Mapped["CreatorAccount"] = relationship(back_populates="notes")
__table_args__ = (
# 同一个时点同一篇只留一行,重复同步不会堆积。
Index("ix_creator_note_stat_unique", "account_id", "note_id", "captured_at", unique=True),
)