feat(schedule): 任务支持「每天定时 / 每周定时」,用选择器而不是手写 cron
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s

- schedule.py: 新增调度计算模块(纯函数,便于单测)。三种模式:interval(每 N 分钟)/ daily(选钟点)/ weekly(选星期 + 钟点)
- 钟点模式是「固定时刻」而非「固定延迟」——从日历重算,所以某轮跑晚了不会把之后每一轮都拖晚。interval 保持原语义:从上一轮开始计时
- 抖动只加给 interval。给「每天 9:00」也加抖动就成了 9:00–9:01 随机触发,操作者选的时间被悄悄改掉,只会像 bug
- 模型 / schema / service: 新增 schedule_mode / schedule_hours / schedule_days / schedule_minute。时钟字段存逗号分隔文本——几个小整数、永远整体读写,单开一张表只会换来 join。interval_minutes 保留且仍是默认值,已有任务不受影响
- service: 改动任何调度字段都按合并后的状态重算 next_run_at。重新启用也算改动,否则停用一个月再打开会带着一个月前的 next_run_at,一保存就立即触发
- 前端: 运行方式三选一 + 小时/星期胶囊多选 + 分钟下拉,并实时预览结果句子。任务卡片改显示后端拼好的 schedule_label,避免列表和编辑器对同一计划给出两种说法
- 校验: 钟点模式至少选一个时间,按周至少选一个星期

tests/test_schedule.py 新增 24 个用例,含「恰好等于当前时刻的档位归属下一天」这个会让调度器自循环的边界。
This commit is contained in:
2026-10-07 15:33:17 +08:00
parent 242e3a7837
commit 8f4e5586e9
10 changed files with 776 additions and 22 deletions
+16 -1
View File
@@ -75,7 +75,7 @@ MODE_NOTE = "note"
class MonitorTask(MonitorBase):
"""One monitored schedule: a set of targets plus an interval."""
"""One monitored schedule: a set of targets, plus when to run them."""
__tablename__ = "monitor_task"
@@ -86,6 +86,21 @@ class MonitorTask(MonitorBase):
enabled: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True)
interval_minutes: Mapped[int] = mapped_column(Integer, nullable=False, default=360)
# How the task is scheduled. `interval` is the original "every N minutes" and
# stays the default; `daily` and `weekly` fire at chosen clock times instead
# (the arithmetic lives in schedule.py).
#
# The clock fields are comma-separated text rather than a child table: they
# are a handful of small integers, always read as a whole, and a table would
# buy nothing but joins.
schedule_mode: Mapped[str] = mapped_column(String(16), nullable=False, default="interval")
# 0-23, e.g. "9,12,18". Empty in interval mode.
schedule_hours: Mapped[str] = mapped_column(String(96), nullable=False, default="")
# 0-6 with Monday = 0, matching Python's date.weekday(). Weekly mode only.
schedule_days: Mapped[str] = mapped_column(String(32), nullable=False, default="")
# Minute past the hour, shared by every time in the schedule.
schedule_minute: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
# Crawl window knobs, mirrored onto each run's CLI flags.
max_notes_count: Mapped[int] = mapped_column(Integer, nullable=False, default=20)
enable_comments: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True)
+167
View File
@@ -0,0 +1,167 @@
# -*- coding: utf-8 -*-
# Copyright (c) 2025 [email protected]
#
# This file is part of MediaCrawler project.
# Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/api/monitor/schedule.py
# GitHub: https://github.com/NanmiCoder
# Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1
#
# 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则:
# 1. 不得用于任何商业用途。
# 2. 使用时应遵守目标平台的使用条款和robots.txt规则。
# 3. 不得进行大规模爬取或对平台造成运营干扰。
# 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。
# 5. 不得用于任何非法或不当的用途。
#
# 详细许可条款请参阅项目根目录下的LICENSE文件。
# 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。
"""Monitor task schedule arithmetic.
Three modes, all expressible by a picker. A raw cron string was ruled out on
purpose -- it is a small language, and the operator should not have to write one
to say "every day at nine":
* ``interval`` -- every N minutes.
* ``daily`` -- at chosen clock times, e.g. 09:00 and 18:30.
* ``weekly`` -- at chosen clock times on chosen weekdays, e.g. Mon-Fri 10:00.
The two clock modes are **fixed-time**, unlike ``interval``, which is fixed-delay.
The distinction matters: for an interval, measuring the next slot from when the
run starts is what stops a slow run from firing back-to-back. For a clock
schedule it would be wrong, because a run that starts at 09:07 would drag every
later run seven minutes late, compounding all day.
Fixed-time also means **no jitter is applied** to clock schedules. The operator
picked a time; quietly running at 09:04 instead of 09:00 is not a feature, it just
looks like a bug. ``interval`` keeps its jitter, where there is no stated time to
contradict.
All arithmetic is in the server's local timezone -- naive datetimes on purpose,
because the container is pinned to the operator's zone via TZ and pretending
otherwise would add a timezone concept nobody asked for.
"""
from datetime import datetime, time, timedelta
from typing import Optional, Sequence
MODE_INTERVAL = "interval"
MODE_DAILY = "daily"
MODE_WEEKLY = "weekly"
SCHEDULE_MODES = (MODE_INTERVAL, MODE_DAILY, MODE_WEEKLY)
CLOCK_MODES = (MODE_DAILY, MODE_WEEKLY)
_MS_PER_MINUTE = 60_000
_WEEKDAY_NAMES = "一二三四五六日"
def parse_hours(raw: Optional[str]) -> list[int]:
"""``"9,18"`` -> ``[9, 18]``. Sorted, de-duplicated, junk dropped."""
return _parse_int_list(raw, 0, 23)
def parse_days(raw: Optional[str]) -> list[int]:
"""``"0,2,4"`` -> ``[0, 2, 4]``. **0 is Monday**, matching ``date.weekday()``."""
return _parse_int_list(raw, 0, 6)
def _parse_int_list(raw: Optional[str], low: int, high: int) -> list[int]:
values: set[int] = set()
for chunk in (raw or "").split(","):
chunk = chunk.strip()
if not chunk:
continue
try:
number = int(chunk)
except ValueError:
# Stored values come from our own UI, but a hand-edited row must not
# be able to crash the scheduler loop.
continue
if low <= number <= high:
values.add(number)
return sorted(values)
def format_hours(hours: Sequence[int]) -> str:
return ",".join(str(hour) for hour in sorted(set(hours)))
def format_days(days: Sequence[int]) -> str:
return ",".join(str(day) for day in sorted(set(days)))
def describe(
*,
mode: str,
interval_minutes: int,
hours: Sequence[int],
days: Sequence[int],
minute: int,
) -> str:
"""One human sentence for the task card.
Lives here rather than in the frontend so the list view and the editor cannot
drift apart on what a schedule means.
"""
if mode == MODE_INTERVAL:
if interval_minutes % 1440 == 0:
return f"每 {interval_minutes // 1440} 天"
if interval_minutes % 60 == 0:
return f"每 {interval_minutes // 60} 小时"
return f"每 {interval_minutes} 分钟"
if not hours:
return "未设置时间"
clock = "、".join(f"{hour:02d}:{minute:02d}" for hour in sorted(set(hours)))
if mode == MODE_DAILY:
return f"每天 {clock}"
if not days:
return f"每天 {clock}"
labels = "、".join(f"周{_WEEKDAY_NAMES[day]}" for day in sorted(set(days)))
return f"{labels} {clock}"
def next_occurrence(
*,
mode: str,
interval_minutes: int,
hours: Sequence[int],
days: Sequence[int],
minute: int,
after_ms: int,
) -> Optional[int]:
"""The next fire time strictly after ``after_ms``, as epoch milliseconds.
``None`` means the schedule can never fire -- a clock mode with no hours
chosen. Callers store that as "no next run" rather than something in the past,
which would otherwise leave the task permanently due and re-running on every
tick.
"""
if mode == MODE_INTERVAL:
return after_ms + max(1, interval_minutes) * _MS_PER_MINUTE
if not hours:
return None
now = datetime.fromtimestamp(after_ms / 1000)
# No weekdays chosen means every day, matching describe(). Without the `days`
# guard an empty selection would produce an empty allowed set, no matching day,
# and a task that silently never runs.
allowed_days = set(days) if (mode == MODE_WEEKLY and days) else set(range(7))
# Eight days of lookahead covers today's remaining slots plus a full week,
# which is more than any weekday selection can need.
for offset in range(8):
day = (now + timedelta(days=offset)).date()
if day.weekday() not in allowed_days:
continue
for hour in sorted(set(hours)):
candidate = datetime.combine(day, time(hour=hour, minute=minute))
if candidate > now:
return int(candidate.timestamp() * 1000)
return None
+38 -9
View File
@@ -23,8 +23,15 @@ is enough here: there is exactly one process, one global crawler subprocess, and
therefore no concurrency to coordinate -- a cron-style library would add a
dependency without adding a capability.
Scheduling is **fixed-delay**, not fixed-rate: ``next_run_at`` is set from the
moment a run starts, so a slow run cannot make its task fire back-to-back.
Two families of schedule, and the difference matters:
* ``interval`` is **fixed-delay**, not fixed-rate -- ``next_run_at`` is measured
from the moment a run starts, so a slow run cannot make its task fire
back-to-back.
* the clock modes (``daily``/``weekly``) are **fixed-time** -- recomputed from the
calendar, so a run that starts late does not drag every later run with it.
The arithmetic for both lives in schedule.py.
"""
import asyncio
@@ -37,7 +44,7 @@ from sqlalchemy import select
from tools.time_util import get_current_timestamp
from ..services import crawler_manager
from . import app_settings
from . import app_settings, schedule
from .db import get_session
from .models import MonitorRun, MonitorTask, RUN_INTERRUPTED, RUN_RUNNING
from .runner import execute_task
@@ -45,10 +52,9 @@ from .settings import get_cookie
POLL_INTERVAL_SECONDS = 20
# Spread tasks sharing an interval so they do not all come due on the same tick.
# Applied to interval mode only -- see the advance step below.
JITTER_SECONDS = 60
_MS_PER_MINUTE = 60_000
class MonitorScheduler:
"""Polls the task table and runs whatever is due."""
@@ -168,11 +174,34 @@ class MonitorScheduler:
# Advance before running so a crash mid-run cannot cause an immediate
# re-fire, and so a long outage coalesces into a single run instead
# of one run per missed interval.
task.next_run_at = (
get_current_timestamp()
+ task.interval_minutes * _MS_PER_MINUTE
+ random.randint(0, JITTER_SECONDS) * 1000
now = get_current_timestamp()
following = schedule.next_occurrence(
mode=task.schedule_mode,
interval_minutes=task.interval_minutes,
hours=schedule.parse_hours(task.schedule_hours),
days=schedule.parse_days(task.schedule_days),
minute=task.schedule_minute,
after_ms=now,
)
if following is None:
# A clock schedule with no times can never fire. The API rejects
# that shape, so this guards against a hand-edited row: park the
# task with no next run rather than leaving it permanently due and
# re-running it on every tick.
task.next_run_at = None
print(
f"[monitor.scheduler] task {task.id} has no usable schedule "
f"and will not run until one is set"
)
elif task.schedule_mode == schedule.MODE_INTERVAL:
# Jitter belongs to the interval mode only. Spreading identical
# intervals apart is the point; nudging a time the operator
# explicitly picked is not -- it just looks like a broken clock.
task.next_run_at = following + random.randint(0, JITTER_SECONDS) * 1000
else:
task.next_run_at = following
task_id = task.id
try:
+93 -5
View File
@@ -28,7 +28,7 @@ from sqlalchemy.ext.asyncio import AsyncSession
from tools.time_util import get_current_timestamp
from . import app_settings, platforms
from . import app_settings, platforms, schedule
from .db import get_session
from .platforms import PLATFORM_XHS
from .models import (
@@ -53,6 +53,28 @@ _background_runs: set[asyncio.Task] = set()
MIN_INTERVAL_MINUTES = 30
MAX_INTERVAL_MINUTES = 7 * 24 * 60
# Changing any of these invalidates the pending run slot.
SCHEDULE_FIELDS = {
"interval_minutes",
"schedule_mode",
"schedule_hours",
"schedule_days",
"schedule_minute",
}
def _require_clock_fields(mode: str, hours: list, days: list) -> None:
"""A clock schedule with no clock time can never fire.
The create schema already rejects that shape, but the update path merges
partial fields and therefore has no schema-level view of the result -- so the
check lives here, where both paths meet.
"""
if mode in schedule.CLOCK_MODES and not hours:
raise ValueError("按钟点调度至少要选一个时间")
if mode == schedule.MODE_WEEKLY and not days:
raise ValueError("按周调度至少要选一个星期")
_CREATOR_URL_RE = re.compile(r"xiaohongshu\.com/user/profile/([A-Za-z0-9_-]+)")
_NOTE_URL_RE = re.compile(r"xiaohongshu\.com/(?:explore|discovery/item)/([A-Za-z0-9_-]+)")
# XHS user ids and note ids are 24-char hex; allow a slightly wider range so a
@@ -139,7 +161,12 @@ async def create_task(session: AsyncSession, payload: Dict[str, Any]) -> Monitor
# the Settings page actually governs new tasks.
defaults = await app_settings.defaults(session, platform)
interval_minutes = payload.get("interval_minutes") or defaults["interval_minutes"]
interval_ms = int(interval_minutes) * 60_000
schedule_mode = payload.get("schedule_mode") or schedule.MODE_INTERVAL
schedule_hours = list(payload.get("schedule_hours") or [])
schedule_days = list(payload.get("schedule_days") or [])
schedule_minute = int(payload.get("schedule_minute") or 0)
_require_clock_fields(schedule_mode, schedule_hours, schedule_days)
task = MonitorTask(
name=payload["name"],
@@ -147,12 +174,23 @@ async def create_task(session: AsyncSession, payload: Dict[str, Any]) -> Monitor
mode=mode,
enabled=payload.get("enabled", True),
interval_minutes=interval_minutes,
schedule_mode=schedule_mode,
schedule_hours=schedule.format_hours(schedule_hours),
schedule_days=schedule.format_days(schedule_days),
schedule_minute=schedule_minute,
max_notes_count=payload.get("max_notes_count") or defaults["max_notes_count"],
enable_comments=payload.get("enable_comments", True),
max_comments_count=payload.get("max_comments_count") or defaults["max_comments_count"],
run_timeout_seconds=payload.get("run_timeout_seconds", 3600),
notify_enabled=payload.get("notify_enabled", False),
next_run_at=now + interval_ms,
next_run_at=schedule.next_occurrence(
mode=schedule_mode,
interval_minutes=interval_minutes,
hours=schedule_hours,
days=schedule_days,
minute=schedule_minute,
after_ms=now,
),
last_status="idle",
created_at=now,
updated_at=now,
@@ -189,10 +227,14 @@ async def update_task(session: AsyncSession, task_id: int, payload: Dict[str, An
if task is None:
raise ValueError(f"Task {task_id} not found")
was_enabled = task.enabled
for field in (
"name",
"enabled",
"interval_minutes",
"schedule_mode",
"schedule_minute",
"max_notes_count",
"enable_comments",
"max_comments_count",
@@ -202,6 +244,13 @@ async def update_task(session: AsyncSession, task_id: int, payload: Dict[str, An
if field in payload and payload[field] is not None:
setattr(task, field, payload[field])
# The clock lists are stored as comma-separated text, so they cannot go
# through the generic loop above.
if payload.get("schedule_hours") is not None:
task.schedule_hours = schedule.format_hours(payload["schedule_hours"])
if payload.get("schedule_days") is not None:
task.schedule_days = schedule.format_days(payload["schedule_days"])
# Replacing targets resets the baseline implicitly: a note set that now
# includes new ids will simply report them as new on the next run.
if payload.get("targets") is not None:
@@ -227,8 +276,23 @@ async def update_task(session: AsyncSession, task_id: int, payload: Dict[str, An
)
)
if "interval_minutes" in payload and payload["interval_minutes"]:
task.next_run_at = get_current_timestamp() + payload["interval_minutes"] * 60_000
# Any change to when the task runs invalidates the pending slot, so recompute
# it from the merged state rather than working out which field moved.
# Re-enabling counts as a change too: otherwise a task switched off for a
# month comes back holding a next_run_at a month in the past and fires the
# instant it is saved.
if (SCHEDULE_FIELDS & set(payload)) or (task.enabled and not was_enabled):
hours = schedule.parse_hours(task.schedule_hours)
days = schedule.parse_days(task.schedule_days)
_require_clock_fields(task.schedule_mode, hours, days)
task.next_run_at = schedule.next_occurrence(
mode=task.schedule_mode,
interval_minutes=task.interval_minutes,
hours=hours,
days=days,
minute=task.schedule_minute,
after_ms=get_current_timestamp(),
)
task.updated_at = get_current_timestamp()
await session.flush()
@@ -623,6 +687,7 @@ async def list_tasks(
"mode": task.mode,
"enabled": task.enabled,
"interval_minutes": task.interval_minutes,
**_schedule_fields(task),
"max_notes_count": task.max_notes_count,
"enable_comments": task.enable_comments,
"max_comments_count": task.max_comments_count,
@@ -644,6 +709,29 @@ async def list_tasks(
]
def _schedule_fields(task: MonitorTask) -> Dict[str, Any]:
"""The schedule columns, plus the sentence the task list renders.
The label is composed here rather than in the frontend so the list and the
editor cannot drift on what a given schedule means.
"""
hours = schedule.parse_hours(task.schedule_hours)
days = schedule.parse_days(task.schedule_days)
return {
"schedule_mode": task.schedule_mode,
"schedule_hours": hours,
"schedule_days": days,
"schedule_minute": task.schedule_minute,
"schedule_label": schedule.describe(
mode=task.schedule_mode,
interval_minutes=task.interval_minutes,
hours=hours,
days=days,
minute=task.schedule_minute,
),
}
async def overview(session: AsyncSession, platform: Optional[str] = None) -> Dict[str, Any]:
"""Headline numbers for the dashboard tiles, scoped to one platform."""
now = get_current_timestamp()