feat(monitor): 运行历史要写清楚失败原因,不能只写「退出码 1」
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s

用户的要求:运行历史的说明要写清楚。现在确实写不清楚 —— 抖音那次失败,运行历史里
只有一句 `Crawler exited with code 1`,而真正的报错 `DataFetchError: account blocked`
埋在子进程的 stderr 里,谁也看不到。

那两者本来是断开的两条路:子进程的输出只流向日志 WebSocket(前端 Terminal 看得到),
而监控层调 run_and_wait() 只拿得到一个退出码。

* crawler_manager 在 _push_log() 里留一份输出尾巴(80 行,每次 start 清空)——
  那是所有输出的唯一出口,挂这儿不会漏。新增 get_output_tail()。
* ingest 新增 diagnose_failure():倒着找第一行像异常的行(traceback 的末行),
  认不出就退回最后一行;管理器自己补的「Crawler exited with code」不是原因,排除掉。
* describe_exit_code() 接受这个原因并附在消息里;失败事件的标题也带上,这样企业微信
  通知和事件流不用翻日志就能看懂。
* runner 把尾巴交给 ingest;「超时/没起来」那条分支同样带上原因 —— -1 同时代表两种
  情况,而要查的东西完全不同。
* 运行历史那一格是截断的(240px),而失败原因现在有一整行 —— 补上 title 悬停显示,
  并放宽到 320px。没有悬停提示等于把最要紧的半句藏起来。

测试 +5:能挑出异常行、不会把管理器自己的话当成原因、没有输出时不报错、认不出时退回
最后一行、以及失败运行同时记下退出码与真因(含事件标题)。
This commit is contained in:
2026-10-10 16:14:54 +08:00
parent 1118d466be
commit e4affe9170
5 changed files with 148 additions and 13 deletions
+51 -8
View File
@@ -38,7 +38,7 @@ import json
import re
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Dict, List, Optional
from typing import Any, Dict, List, Optional, Sequence
from sqlalchemy import func, select
from sqlalchemy.ext.asyncio import AsyncSession
@@ -122,13 +122,46 @@ _WINDOWS_EXIT_REASONS = {
}
def describe_exit_code(code: int) -> str:
"""Render an exit code so a human can act on it."""
def describe_exit_code(code: int, cause: Optional[str] = None) -> str:
"""Render an exit code so a human can act on it.
只有退出码时信息量约等于零(`code 1` 什么都能是),所以把从输出尾巴里认出来的
异常一并附上 —— 运行历史里那一格显示的正是这句话。
"""
unsigned = code & 0xFFFFFFFF if code < 0 else code
reason = _WINDOWS_EXIT_REASONS.get(unsigned)
base = f"Crawler exited with code {code}"
if reason:
return f"Crawler exited with code {code} (0x{unsigned:08X}): {reason}"
return f"Crawler exited with code {code}"
base = f"{base} (0x{unsigned:08X}): {reason}"
return f"{base};原因:{cause}" if cause else base
# 从爬虫输出里认出一行「异常」。Python 的 traceback 末行形如
# ``media_platform.douyin.exception.DataFetchError: account blocked``。
_EXCEPTION_LINE_RE = re.compile(r"^[\w.]*[A-Za-z](?:Error|Exception|Timeout)\b")
def diagnose_failure(output_tail: Optional[Sequence[str]]) -> Optional[str]:
"""从爬虫输出的末尾挑出最能说明问题的一行。
「退出码 1」等于什么都没说:真正的报错埋在子进程的 stderr 里。倒着找第一行看起来
像异常的行(traceback 的末行),找不到就退回最后一行有效输出。
"""
if not output_tail:
return None
lines = [line.strip() for line in output_tail if line and line.strip()]
# 管理器自己补的那两句不是爬虫的报错,别被当成失败原因。
noise = ("Crawler exited with code", "Crawler completed successfully")
lines = [line for line in lines if not line.startswith(noise)]
if not lines:
return None
for line in reversed(lines):
if _EXCEPTION_LINE_RE.match(line):
return line[:300]
return lines[-1][:300]
@dataclass
@@ -558,6 +591,7 @@ async def ingest_run(
run: MonitorRun,
task: MonitorTask,
out_dir: Path,
output_tail: Optional[Sequence[str]] = None,
) -> IngestResult:
"""Ingest one finished run and return what changed.
@@ -568,14 +602,23 @@ async def ingest_run(
# A non-zero exit is a genuine crash: trust nothing this run produced.
if run.exit_code not in (0, None):
run.status = RUN_FAILED
run.error_message = describe_exit_code(run.exit_code)
cause = diagnose_failure(output_tail)
run.error_message = describe_exit_code(run.exit_code, cause)
title = f"采集进程异常退出(code={run.exit_code})"
if cause:
# 标题里也带上真因:企业微信通知和事件流都只看这一行,不写就还得去翻日志。
title = f"{title}:{cause}"
await _emit(
session,
run,
EVENT_RUN_FAILED,
f"采集进程异常退出(code={run.exit_code})",
title,
severity="error",
payload={"exit_code": run.exit_code, "detail": run.error_message},
payload={
"exit_code": run.exit_code,
"detail": run.error_message,
"cause": cause,
},
)
return IngestResult(status=RUN_FAILED, error=run.error_message)
+11 -2
View File
@@ -40,7 +40,7 @@ from ..schemas import (
from ..services import crawler_manager
from . import adapters, app_settings, covers, notify
from .db import get_session
from .ingest import IngestResult, ingest_run
from .ingest import IngestResult, diagnose_failure, ingest_run
from .models import (
MODE_CREATOR,
RUN_FAILED,
@@ -259,11 +259,20 @@ async def execute_task(task_id: int, trigger: str = "manual") -> IngestResult:
run.finished_at = get_current_timestamp()
run.exit_code = exit_code
run.error_message = "Run was killed by timeout or failed to start"
# -1 同时代表「超时」和「根本没起来」,两者要查的东西完全不同。输出尾巴里
# 有异常就带上它,否则运行历史里只能看到这句没有信息量的话。
cause = diagnose_failure(crawler_manager.get_output_tail())
if cause:
run.error_message = f"{run.error_message};原因:{cause}"
result = IngestResult(status=RUN_TIMEOUT, error=run.error_message)
else:
run.exit_code = exit_code
run.finished_at = get_current_timestamp()
result = await ingest_run(session, run, task, out_dir)
# 把爬虫输出的尾巴交给 ingest:退出码本身说明不了问题,运行历史里要显示的
# 是真正的报错(比如抖音的 DataFetchError: account blocked)。
result = await ingest_run(
session, run, task, out_dir, output_tail=crawler_manager.get_output_tail()
)
# A run that authenticated fine is the only useful signal that the
# stored cookie still works.
+22 -1
View File
@@ -20,13 +20,20 @@ import asyncio
import subprocess
import signal
import os
from typing import Optional, List
from collections import deque
from typing import Deque, Optional, List
from datetime import datetime
from pathlib import Path
from ..schemas import CrawlerStartRequest, LogEntry
from .interpreter import resolve_python_cmd
# 留住多少行爬虫输出,供 run_and_wait 的调用方诊断失败原因。
# 子进程的输出本来只流向日志 WebSocket,监控层只看得到退出码 —— 于是「退出码 1」
# 成了运行历史里唯一的信息,真正的报错(比如抖音的 `DataFetchError: account blocked`)
# 谁也看不到。留个尾巴,让失败原因能被写进 run.error_message。
OUTPUT_TAIL_LINES = 80
class CrawlerManager:
"""Crawler process manager"""
@@ -49,6 +56,15 @@ class CrawlerManager:
# by any concurrent start(), so waiters need an explicit event instead.
self._done: asyncio.Event = asyncio.Event()
self.last_exit_code: Optional[int] = None
# 本次运行输出的末尾若干行。见 OUTPUT_TAIL_LINES。
self._output_tail: Deque[str] = deque(maxlen=OUTPUT_TAIL_LINES)
def get_output_tail(self) -> List[str]:
"""最近一次运行的输出尾巴(最早的排前面)。
只在 run_and_wait() 返回之后读才有意义 —— 它等到读输出的任务收尾才唤醒。
"""
return list(self._output_tail)
@property
def logs(self) -> List[LogEntry]:
@@ -114,6 +130,9 @@ class CrawlerManager:
async def _push_log(self, entry: LogEntry):
"""Push log to queue"""
# 这里是所有输出的唯一出口(读循环、收尾、以及管理器自己的提示都走它),
# 所以尾巴挂在这儿最省事,也不会漏。
self._output_tail.append(entry.message)
if self._log_queue is not None:
try:
self._log_queue.put_nowait(entry)
@@ -149,6 +168,8 @@ class CrawlerManager:
# Reset completion signalling for this run
self._done.clear()
self.last_exit_code = None
# 尾巴只属于本次运行,否则上一轮的报错会混进这一轮的诊断里。
self._output_tail.clear()
# Clear pending queue (don't replace object to avoid WebSocket broadcast coroutine holding old queue reference)
if self._log_queue is None: