From 2b9ebdad87168595dc9cb85f9f12fd7fdfc2d99b Mon Sep 17 00:00:00 2001 From: butubb <1422726308@qq.com> Date: Wed, 7 Oct 2026 15:46:11 +0800 Subject: [PATCH] =?UTF-8?q?fix(ui):=20=E3=80=8C=E6=AF=8F=E8=BD=AE=E6=9C=80?= =?UTF-8?q?=E5=A4=9A=E9=87=87=E9=9B=86=E4=BD=9C=E5=93=81=E6=95=B0=E3=80=8D?= =?UTF-8?q?=E6=A0=87=E7=AD=BE=E6=98=AF=E9=94=99=E7=9A=84=E2=80=94=E2=80=94?= =?UTF-8?q?=E5=AE=83=E6=98=AF=E6=AF=8F=E4=B8=AA=E5=8D=9A=E4=B8=BB=E7=9A=84?= =?UTF-8?q?=E4=B8=8A=E9=99=90?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 爬虫里这个值是在 per-creator 的函数内比较的(client.py get_all_notes_by_creator 的 result 是局部变量),而外层 for 循环遍历全部目标。所以 100 个目标 × 20 篇 = 单轮最多 2000 篇,一篇都不会被丢弃。标签写成「每轮最多」会让人以为超出的会被截掉。 - creator 模式:标签改为「每个博主最多采集作品数」,并实时算出「N 个目标 × M 篇 → 单轮最多 X 篇」 - note 模式:禁用该输入并说明「此项不生效」——get_specified_notes 里没有任何 CRAWLER_MAX_NOTES_COUNT 引用,列出的每个链接都会被逐条抓 - 单轮估算超过 500 篇时给出警告:每篇还要抓最多 max_comments_count 条评论、并发为 1,容易触发限流,也可能跑不完就被默认 1 小时的任务超时中断 --- .../components/monitor/TaskEditorDialog.tsx | 47 +++++++++++++++++-- 1 file changed, 43 insertions(+), 4 deletions(-) diff --git a/webui/src/components/monitor/TaskEditorDialog.tsx b/webui/src/components/monitor/TaskEditorDialog.tsx index 388360f..2718b12 100644 --- a/webui/src/components/monitor/TaskEditorDialog.tsx +++ b/webui/src/components/monitor/TaskEditorDialog.tsx @@ -57,6 +57,15 @@ const SCHEDULE_MODE_OPTIONS: Array<{ value: ScheduleMode; label: string }> = [ ] const HOURS = Array.from({ length: 24 }, (_, hour) => hour) + +/** + * Above this many works per run, warn about the request volume. + * + * The runner serialises crawls (max_concurrency_num=1) and each note also pulls + * up to `max_comments_count` comments, so the cost is works × (1 + comments) and + * the default run timeout is an hour. + */ +const NOTE_VOLUME_WARN = 500 const MINUTES = Array.from({ length: 60 }, (_, minute) => minute) const WEEKDAYS = ['周一', '周二', '周三', '周四', '周五', '周六', '周日'] const pad = (value: number) => String(value).padStart(2, '0') @@ -371,18 +380,48 @@ export function TaskEditorDialog({ open, onOpenChange, task }: TaskEditorDialogP
setMaxNotes(event.target.value)} + disabled={mode === 'note'} className="h-9 text-xs" /> -

- 只取最新的前 N 条,决定了"该博主的作品"覆盖范围 -

+ + {mode === 'creator' ? ( + <> + {/* 这是「每个博主」的上限,不是一轮的总量 —— 爬虫里这个值是在 + per-creator 的函数内比较的(client.py get_all_notes_by_creator), + 而外层 for 循环会遍历全部目标。所以 100 个目标 × 20 篇 = 单轮 + 最多 2000 篇。标签写成「每轮最多」会让人以为超出的会被丢弃。 */} +

+ 这是每个博主的上限, + 不是一轮的总量。只取该博主最新的前 N 条,超出的不会补抓。 +

+

+ {targetList.length} 个目标 × {maxNotes} 篇 × 每人 1 次 + → 单轮最多{' '} + + {targetList.length * (Number(maxNotes) || 0)} + {' '} + 篇 +

+ {targetList.length * (Number(maxNotes) || 0) > NOTE_VOLUME_WARN && ( +

+ 单轮量偏大:每篇还要抓最多 {maxComments} 条评论,且并发为 1。 + 容易触发平台限流,也可能跑不完就被任务超时(默认 1 小时)中断。 + 建议调低这个数,或拆成几个任务。 +

+ )} + + ) : ( +

+ 笔记模式下此项不生效:你列出的每个链接都会被逐条抓取。 +

+ )}