diff --git a/api/monitor/ingest.py b/api/monitor/ingest.py index d110404..79428f4 100644 --- a/api/monitor/ingest.py +++ b/api/monitor/ingest.py @@ -38,7 +38,7 @@ import json import re from dataclasses import dataclass, field from pathlib import Path -from typing import Any, Dict, List, Optional +from typing import Any, Dict, List, Optional, Sequence from sqlalchemy import func, select from sqlalchemy.ext.asyncio import AsyncSession @@ -122,13 +122,46 @@ _WINDOWS_EXIT_REASONS = { } -def describe_exit_code(code: int) -> str: - """Render an exit code so a human can act on it.""" +def describe_exit_code(code: int, cause: Optional[str] = None) -> str: + """Render an exit code so a human can act on it. + + 只有退出码时信息量约等于零(`code 1` 什么都能是),所以把从输出尾巴里认出来的 + 异常一并附上 —— 运行历史里那一格显示的正是这句话。 + """ unsigned = code & 0xFFFFFFFF if code < 0 else code reason = _WINDOWS_EXIT_REASONS.get(unsigned) + base = f"Crawler exited with code {code}" if reason: - return f"Crawler exited with code {code} (0x{unsigned:08X}): {reason}" - return f"Crawler exited with code {code}" + base = f"{base} (0x{unsigned:08X}): {reason}" + return f"{base};原因:{cause}" if cause else base + + +# 从爬虫输出里认出一行「异常」。Python 的 traceback 末行形如 +# ``media_platform.douyin.exception.DataFetchError: account blocked``。 +_EXCEPTION_LINE_RE = re.compile(r"^[\w.]*[A-Za-z](?:Error|Exception|Timeout)\b") + + +def diagnose_failure(output_tail: Optional[Sequence[str]]) -> Optional[str]: + """从爬虫输出的末尾挑出最能说明问题的一行。 + + 「退出码 1」等于什么都没说:真正的报错埋在子进程的 stderr 里。倒着找第一行看起来 + 像异常的行(traceback 的末行),找不到就退回最后一行有效输出。 + """ + if not output_tail: + return None + + lines = [line.strip() for line in output_tail if line and line.strip()] + # 管理器自己补的那两句不是爬虫的报错,别被当成失败原因。 + noise = ("Crawler exited with code", "Crawler completed successfully") + lines = [line for line in lines if not line.startswith(noise)] + if not lines: + return None + + for line in reversed(lines): + if _EXCEPTION_LINE_RE.match(line): + return line[:300] + + return lines[-1][:300] @dataclass @@ -558,6 +591,7 @@ async def ingest_run( run: MonitorRun, task: MonitorTask, out_dir: Path, + output_tail: Optional[Sequence[str]] = None, ) -> IngestResult: """Ingest one finished run and return what changed. @@ -568,14 +602,23 @@ async def ingest_run( # A non-zero exit is a genuine crash: trust nothing this run produced. if run.exit_code not in (0, None): run.status = RUN_FAILED - run.error_message = describe_exit_code(run.exit_code) + cause = diagnose_failure(output_tail) + run.error_message = describe_exit_code(run.exit_code, cause) + title = f"采集进程异常退出(code={run.exit_code})" + if cause: + # 标题里也带上真因:企业微信通知和事件流都只看这一行,不写就还得去翻日志。 + title = f"{title}:{cause}" await _emit( session, run, EVENT_RUN_FAILED, - f"采集进程异常退出(code={run.exit_code})", + title, severity="error", - payload={"exit_code": run.exit_code, "detail": run.error_message}, + payload={ + "exit_code": run.exit_code, + "detail": run.error_message, + "cause": cause, + }, ) return IngestResult(status=RUN_FAILED, error=run.error_message) diff --git a/api/monitor/runner.py b/api/monitor/runner.py index ec72b7e..3e6def1 100644 --- a/api/monitor/runner.py +++ b/api/monitor/runner.py @@ -40,7 +40,7 @@ from ..schemas import ( from ..services import crawler_manager from . import adapters, app_settings, covers, notify from .db import get_session -from .ingest import IngestResult, ingest_run +from .ingest import IngestResult, diagnose_failure, ingest_run from .models import ( MODE_CREATOR, RUN_FAILED, @@ -259,11 +259,20 @@ async def execute_task(task_id: int, trigger: str = "manual") -> IngestResult: run.finished_at = get_current_timestamp() run.exit_code = exit_code run.error_message = "Run was killed by timeout or failed to start" + # -1 同时代表「超时」和「根本没起来」,两者要查的东西完全不同。输出尾巴里 + # 有异常就带上它,否则运行历史里只能看到这句没有信息量的话。 + cause = diagnose_failure(crawler_manager.get_output_tail()) + if cause: + run.error_message = f"{run.error_message};原因:{cause}" result = IngestResult(status=RUN_TIMEOUT, error=run.error_message) else: run.exit_code = exit_code run.finished_at = get_current_timestamp() - result = await ingest_run(session, run, task, out_dir) + # 把爬虫输出的尾巴交给 ingest:退出码本身说明不了问题,运行历史里要显示的 + # 是真正的报错(比如抖音的 DataFetchError: account blocked)。 + result = await ingest_run( + session, run, task, out_dir, output_tail=crawler_manager.get_output_tail() + ) # A run that authenticated fine is the only useful signal that the # stored cookie still works. diff --git a/api/services/crawler_manager.py b/api/services/crawler_manager.py index 95138c5..3869ca8 100644 --- a/api/services/crawler_manager.py +++ b/api/services/crawler_manager.py @@ -20,13 +20,20 @@ import asyncio import subprocess import signal import os -from typing import Optional, List +from collections import deque +from typing import Deque, Optional, List from datetime import datetime from pathlib import Path from ..schemas import CrawlerStartRequest, LogEntry from .interpreter import resolve_python_cmd +# 留住多少行爬虫输出,供 run_and_wait 的调用方诊断失败原因。 +# 子进程的输出本来只流向日志 WebSocket,监控层只看得到退出码 —— 于是「退出码 1」 +# 成了运行历史里唯一的信息,真正的报错(比如抖音的 `DataFetchError: account blocked`) +# 谁也看不到。留个尾巴,让失败原因能被写进 run.error_message。 +OUTPUT_TAIL_LINES = 80 + class CrawlerManager: """Crawler process manager""" @@ -49,6 +56,15 @@ class CrawlerManager: # by any concurrent start(), so waiters need an explicit event instead. self._done: asyncio.Event = asyncio.Event() self.last_exit_code: Optional[int] = None + # 本次运行输出的末尾若干行。见 OUTPUT_TAIL_LINES。 + self._output_tail: Deque[str] = deque(maxlen=OUTPUT_TAIL_LINES) + + def get_output_tail(self) -> List[str]: + """最近一次运行的输出尾巴(最早的排前面)。 + + 只在 run_and_wait() 返回之后读才有意义 —— 它等到读输出的任务收尾才唤醒。 + """ + return list(self._output_tail) @property def logs(self) -> List[LogEntry]: @@ -114,6 +130,9 @@ class CrawlerManager: async def _push_log(self, entry: LogEntry): """Push log to queue""" + # 这里是所有输出的唯一出口(读循环、收尾、以及管理器自己的提示都走它), + # 所以尾巴挂在这儿最省事,也不会漏。 + self._output_tail.append(entry.message) if self._log_queue is not None: try: self._log_queue.put_nowait(entry) @@ -149,6 +168,8 @@ class CrawlerManager: # Reset completion signalling for this run self._done.clear() self.last_exit_code = None + # 尾巴只属于本次运行,否则上一轮的报错会混进这一轮的诊断里。 + self._output_tail.clear() # Clear pending queue (don't replace object to avoid WebSocket broadcast coroutine holding old queue reference) if self._log_queue is None: diff --git a/tests/test_monitor_ingest.py b/tests/test_monitor_ingest.py index 7905969..7bccd78 100644 --- a/tests/test_monitor_ingest.py +++ b/tests/test_monitor_ingest.py @@ -37,7 +37,12 @@ from sqlalchemy.pool import StaticPool from tools.time_util import get_current_timestamp from api.monitor import adapters -from api.monitor.ingest import describe_exit_code, ingest_run, parse_count +from api.monitor.ingest import ( + describe_exit_code, + diagnose_failure, + ingest_run, + parse_count, +) from api.monitor.models import ( EVENT_AUTH_FAILURE, EVENT_METRIC_DELTA, @@ -750,3 +755,55 @@ class TestNicknameRefresh: assert ( len(list((await db.scalars(select(MonitorComment))).all())) == 1 ) + + +class TestFailureDiagnosis: + """失败原因要能被人看懂。 + + 只写「退出码 1」等于什么都没说 —— 真正的报错埋在子进程的 stderr 里,而运行历史 + 里那一格显示的正是 run.error_message。 + """ + + TAIL = [ + "2026-10-10 15:18:34 MediaCrawler INFO (core.py:385) - [DouYinCrawler] CDP浏览器信息", + "Traceback (most recent call last):", + ' File "/app/main.py", line 114, in main', + " await crawler.start()", + "media_platform.douyin.exception.DataFetchError: account blocked, ", + ] + + def test_the_exception_line_is_picked_out_of_the_tail(self): + assert ( + diagnose_failure(self.TAIL) + == "media_platform.douyin.exception.DataFetchError: account blocked," + ) + + def test_the_managers_own_lines_are_not_mistaken_for_the_cause(self): + """管理器自己补的那两句不是爬虫的报错,别被当成失败原因。""" + assert diagnose_failure(["Crawler exited with code: 1"]) is None + assert diagnose_failure(["Crawler completed successfully"]) is None + + def test_nothing_to_say_is_not_an_error(self): + assert diagnose_failure(None) is None + assert diagnose_failure([]) is None + + def test_it_falls_back_to_the_last_line(self): + assert ( + diagnose_failure(["started fine", "then something odd"]) + == "then something odd" + ) + + @pytest.mark.asyncio + async def test_a_failed_run_records_both_the_code_and_the_cause(self, db, tmp_path): + task = await _make_task(db) + run = await _make_run(db, task, started_at=1, exit_code=1) + + result = await ingest_run(db, run, task, tmp_path, output_tail=self.TAIL) + + assert result.status == RUN_FAILED + # 退出码和真因都要在,缺一个都还得去翻日志。 + assert "code 1" in run.error_message + assert "account blocked" in run.error_message + + events = await _events(db, EVENT_RUN_FAILED) + assert "account blocked" in events[0].title diff --git a/webui/src/components/monitor/RunHistory.tsx b/webui/src/components/monitor/RunHistory.tsx index ff22dd6..a34aa53 100644 --- a/webui/src/components/monitor/RunHistory.tsx +++ b/webui/src/components/monitor/RunHistory.tsx @@ -83,7 +83,12 @@ export function RunHistory({ taskId }: RunHistoryProps) { {formatDateTime(run.started_at)} ({formatRelative(run.started_at)}) - + {/* 这一格是截断的,而失败原因现在会带上一整行异常 —— 没有 title + 就等于把最要紧的那半句藏起来了。 */} + {run.error_message ?? ''}