feat(monitor): 作品栏和评论栏显示作品的发布日期
Deploy VitePress site to Pages / build (push) Canceled after 0s
Deploy VitePress site to Pages / Deploy (push) Canceled after 0s

需求:抖音和小红书都要能看到作品的发布日期。

`MonitorNote.published_at` 其实**一直在库里**(ingest 早就按平台取:小红书 time、
抖音 create_time),只是从来没往 API 和界面上透 —— 后端序列化没这个键,前端类型里
也没有,所以界面上只有「首次发现」。

* 后端:list_notes 的序列化补上 published_at;_note_meta_map 也带上,于是
  list_comments 多一个 note_published_at,分组接口的桶多一个 published_at。
* 前端:NotesTable 新增「发布日期」列(要让分组表头的 colspan 从 +4 变 +5);
  评论栏作品那一层在标题旁显示日期 —— 同名作品不少,日期能帮着认。
* 新增 formatDate:发布日期问的是「哪一天发的」,绝对日期比「3天前」好认,也不会
  每天看都在变。具体到分钟的版本放在 title 里,悬停可见。

刻意和「首次发现」分开:前者是作者发布的那天,后者是我们第一次看到它的那天。把一个
早就存在的作品加进监控时,两者能差好几个月 —— 测试里就用不同的值把这两者钉住。

顺带修正一处过时注释:前端类型里还写着 creator_name 是「已脱敏的昵称」,
脱敏已经在上一个提交里关掉了(config.MASK_NICKNAME)。

测试 +2:桶要带发布日期;作品列表接口要带,且它不等于 first_seen_at。
This commit is contained in:
2026-10-10 16:06:22 +08:00
parent cd85587f00
commit 3486c7f524
7 changed files with 101 additions and 10 deletions
+9 -1
View File
@@ -404,8 +404,14 @@ async def list_notes(
"note_id": note.note_id,
"title": note.title,
"note_url": note.note_url,
# 作品的发布时间(爬虫侧:小红书叫 time、抖音叫 create_time)。
# 和 first_seen_at 不是一回事 —— 那是**我们第一次看到它**的时间;把一个
# 早就存在的作品加进监控时,两者能差好几个月。可能为 null(平台没给,
# 或者值解析不出来),所以前端要能显示成「—」。
"published_at": note.published_at,
# 按博主分组用。creator_hash 是唯一稳定的创作者标识(原始 user_id
# 被爬虫刻意匿名化了),creator_name 是已脱敏的昵称。
# 被爬虫刻意匿名化了),creator_name 是昵称本身 —— 本仓库关掉了脱敏
# (见 config.MASK_NICKNAME),所以就是原文。
"creator_hash": note.creator_hash,
"creator_name": note.creator_name,
# 优先给本地缓存地址:远程地址带签名、会过期(实测隔天即 403),
@@ -488,6 +494,7 @@ async def _note_meta_map(
"note_title": row.title,
"note_cover": covers.cover_url(row.note_id, row.cover),
"note_url": row.note_url,
"published_at": row.published_at,
"task_id": row.task_id,
# 博主维度也带上,评论流才能按 博主 -> 作品 -> 评论 三级展开。
"creator_hash": row.creator_hash,
@@ -539,6 +546,7 @@ async def list_comments(
"note_url": meta.get(row.note_id, {}).get("note_url", ""),
"note_creator_hash": meta.get(row.note_id, {}).get("creator_hash", ""),
"note_creator_name": meta.get(row.note_id, {}).get("creator_name", ""),
"note_published_at": meta.get(row.note_id, {}).get("published_at"),
}
for row in comments
]
+3
View File
@@ -178,6 +178,9 @@ async def list_comments(
# 同一个桶里的评论必然同属一个作品,所以取哪一条都一样。
"creator_hash": comment["note_creator_hash"],
"creator_name": comment["note_creator_name"],
# 作品的发布时间。评论流按 博主 → 作品 → 评论 展开时,作品那一层
# 光有标题不够 —— 同名作品不少,日期能帮着认。
"published_at": comment["note_published_at"],
"comments": [],
},
)
+40 -4
View File
@@ -36,6 +36,11 @@ from api.monitor.models import (
TASK_NAME = "评论归属测试"
# 作品的发布时间。和 first_seen_at(我们第一次看到它)刻意取不同的值 —— 两者混成
# 一个概念是最容易犯的错。
PUBLISHED_A = 1_699_000_000_000
PUBLISHED_B = 1_699_100_000_000
async def _seed():
"""Two works; three comments on the first, one on the second."""
@@ -50,9 +55,9 @@ async def _seed():
await session.flush()
# 两个作品**属于不同的博主** —— 评论流最外层按创作者分组,同一个人就没得测了。
for note_id, title, creator_hash, creator_name in (
("note-a", "作品甲", "hash-a", "博主甲"),
("note-b", "作品乙", "hash-b", "博主乙"),
for note_id, title, creator_hash, creator_name, published_at in (
("note-a", "作品甲", "hash-a", "博主甲", PUBLISHED_A),
("note-b", "作品乙", "hash-b", "博主乙", PUBLISHED_B),
):
session.add(
MonitorNote(
@@ -60,7 +65,7 @@ async def _seed():
note_url=f"https://www.xiaohongshu.com/explore/{note_id}",
cover=f"https://img/{note_id}.jpg", creator_hash=creator_hash,
creator_name=creator_name,
source_kind="video", published_at=None,
source_kind="video", published_at=published_at,
first_seen_run_id=1, first_seen_at=1_700_000_000_000,
last_seen_run_id=1, last_seen_at=1_700_000_000_000,
)
@@ -165,6 +170,37 @@ class TestGroupByNote:
# 两个作品的创作者必须真的不同,否则界面上照样分不出来。
assert groups["note-a"]["creator_hash"] != groups["note-b"]["creator_hash"]
@pytest.mark.asyncio
async def test_each_bucket_carries_the_publish_date(self, client):
"""作品那一层要带发布日期:同名作品不少,日期能帮着认。
注意它和 first_seen_at 是两个概念 —— 前者是作者发布的那天,后者是我们第一次
看到它的那天。把老作品加进监控时两者能差好几个月。
"""
groups = {
group["note_id"]: group
for group in (
await client.get("/api/monitor/comments", params={"group_by": "note"})
).json()["groups"]
}
assert groups["note-a"]["published_at"] == PUBLISHED_A
assert groups["note-b"]["published_at"] == PUBLISHED_B
assert groups["note-a"]["published_at"] != groups["note-b"]["published_at"]
@pytest.mark.asyncio
async def test_the_notes_endpoint_exposes_the_publish_date(self, client):
"""作品列表也要带上它 —— 作品栏就是靠这个显示「发布日期」列的。"""
notes = {
note["note_id"]: note
for note in (await client.get("/api/monitor/notes")).json()["notes"]
}
assert notes["note-a"]["published_at"] == PUBLISHED_A
assert notes["note-b"]["published_at"] == PUBLISHED_B
# 和「首次发现」不是同一个值 —— 两者混了的话这个断言会抓到。
assert notes["note-a"]["first_seen_at"] != notes["note-a"]["published_at"]
@pytest.mark.asyncio
async def test_flat_shape_is_unchanged_without_the_flag(self, client):
body = (await client.get("/api/monitor/comments")).json()
+10 -1
View File
@@ -25,7 +25,7 @@ import {
useMonitorCommentsGrouped,
} from '@/hooks/useMonitor'
import { monitorApi } from '@/lib/api'
import { formatRelative } from '@/lib/monitorFormat'
import { formatDate, formatRelative } from '@/lib/monitorFormat'
import type { CommentBucket, MonitorComment } from '@/types/monitor'
import { NoteCover } from './NoteCover'
@@ -137,6 +137,15 @@ function CollapsibleGroup({
noteId={bucket.note_id}
/>
</div>
{/* 作品那一层光有标题不够 —— 同名的作品不少,发布日期能帮着认。 */}
{bucket.published_at && (
<span
className="text-[10px] font-mono text-cyber-text-muted flex-shrink-0"
title="作品发布时间"
>
{formatDate(bucket.published_at)}
</span>
)}
<Badge variant="outline" className="text-[10px] px-1.5 py-0 flex-shrink-0">
{bucket.comments.length} 条
</Badge>
+18 -3
View File
@@ -3,7 +3,13 @@ import { ChevronDown, ChevronRight, ExternalLink, Users } from 'lucide-react'
import { Badge } from '@/components/ui/badge'
import { useMonitorNotes } from '@/hooks/useMonitor'
import { formatCount, formatDelta, formatRelative } from '@/lib/monitorFormat'
import {
formatCount,
formatDate,
formatDateTime,
formatDelta,
formatRelative,
} from '@/lib/monitorFormat'
import type { MonitorNote, NoteMetrics } from '@/types/monitor'
import { NoteCover } from './NoteCover'
import { NoteTrendChart } from './NoteTrendChart'
@@ -74,8 +80,8 @@ function sumMetrics(notes: MonitorNote[]) {
return { values, deltas }
}
// 多出来的那个空列给展开箭头;分组表头会跨掉整行。
const COLUMN_COUNT = METRIC_COLUMNS.length + 4
// 多出来的那几列:展开箭头、作品、发布日期、首次发现、跳转链接;分组表头会跨掉整行。
const COLUMN_COUNT = METRIC_COLUMNS.length + 5
/**
* 作品表,**按博主分组**。
@@ -140,6 +146,7 @@ export function NotesTable({ taskId, onlyNew }: NotesTableProps) {
{column.label}
</th>
))}
<th className="text-right font-normal py-2 px-2">发布日期</th>
<th className="text-right font-normal py-2 px-2">首次发现</th>
<th className="w-8" />
</tr>
@@ -230,6 +237,14 @@ export function NotesTable({ taskId, onlyNew }: NotesTableProps) {
/>
</td>
))}
{/* 发布日期和「首次发现」是两回事:把早就发过的作品加进监控时,
前者是作者发的那天,后者是我们第一次看到它的那天。 */}
<td
className="py-2 px-2 text-right text-[10px] text-cyber-text-muted whitespace-nowrap"
title={formatDateTime(note.published_at)}
>
{formatDate(note.published_at)}
</td>
<td className="py-2 px-2 text-right text-[10px] text-cyber-text-muted">
{formatRelative(note.first_seen_at)}
</td>
+13
View File
@@ -38,6 +38,19 @@ export function formatDateTime(ms: number | null | undefined): string {
return `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())} ${pad(d.getHours())}:${pad(d.getMinutes())}`
}
/**
* 只到日。
*
* 发布日期问的是「哪一天发的」,绝对日期比「3天前」好认 —— 后者每天看都在变,而且
* 没法跟平台上的日期对。具体到分钟的那份放 title 里,需要时悬停看。
*/
export function formatDate(ms: number | null | undefined): string {
if (!ms) return '—'
const d = new Date(ms)
const pad = (n: number) => String(n).padStart(2, '0')
return `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())}`
}
/** Human interval label for task cards. */
export function formatInterval(minutes: number): string {
if (minutes % 1440 === 0) return `${minutes / 1440} 天`
+8 -1
View File
@@ -91,10 +91,13 @@ export interface MonitorNote {
*
* `creator_hash` 是唯一稳定的分组依据 —— 爬虫刻意不落原始 user_id(见
* tools/user_hash.py),所以没有比它更具体的身份了。
* `creator_name` 是**已脱敏**的昵称(张***三 这种)。
* `creator_name` 是昵称本身(本仓库关掉了脱敏,见 config.MASK_NICKNAME)。
*/
creator_hash: string
creator_name: string
/** 作品的**发布**时间(爬虫侧:小红书 time、抖音 create_time)。可能与
* `first_seen_at` 差很远 —— 后者是「我们第一次看到它」的时间。平台没给时为 null。 */
published_at: number | null
first_seen_at: number
last_seen_at: number
is_new: boolean
@@ -130,6 +133,8 @@ export interface MonitorComment {
/** 所属作品的创作者 —— 评论流按 博主 -> 作品 -> 评论 三级展开时用。 */
note_creator_hash: string
note_creator_name: string
/** 所属作品的发布时间。 */
note_published_at: number | null
}
/** One work that has comments, for the filter dropdown. */
@@ -152,6 +157,8 @@ export interface CommentBucket {
note_url: string
creator_hash: string
creator_name: string
/** 作品的发布时间。 */
published_at: number | null
comments: MonitorComment[]
}