# -*- coding: utf-8 -*- # Copyright (c) 2025 relakkes@gmail.com # # This file is part of MediaCrawler project. # Repository: https://github.com/NanmiCoder/MediaCrawler/blob/main/tests/test_monitor_api.py # GitHub: https://github.com/NanmiCoder # Licensed under NON-COMMERCIAL LEARNING LICENSE 1.1 # # 声明:本代码仅供学习和研究目的使用。使用者应遵守以下原则: # 1. 不得用于任何商业用途。 # 2. 使用时应遵守目标平台的使用条款和robots.txt规则。 # 3. 不得进行大规模爬取或对平台造成运营干扰。 # 4. 应合理控制请求频率,避免给目标平台带来不必要的负担。 # 5. 不得用于任何非法或不当的用途。 # # 详细许可条款请参阅项目根目录下的LICENSE文件。 # 使用本代码即表示您同意遵守上述原则和LICENSE中的所有条款。 """API-level tests for the monitoring endpoints. Run against an ASGI transport with a temporary database, so no server, network or login is required. Lifespan is deliberately not exercised: it would start the scheduler, and these tests only cover routing, validation and persistence. """ import httpx import pytest import pytest_asyncio from api.main import app from api.monitor import db as monitor_db from api.monitor.service import TargetParseError, parse_target_input CREATOR_URL = ( "https://www.xiaohongshu.com/user/profile/5f58bd990000000001003753" "?xsec_token=ABYVg1evluJZZzpMX-VWzchxQ1qSNVW3r-jOEnKqMcgZw=&xsec_source=pc_search" ) NOTE_URL = "https://www.xiaohongshu.com/explore/6aa3d827000000002802c5c8?xsec_token=TOKEN&xsec_source=pc_search" @pytest_asyncio.fixture async def client(tmp_path): monitor_db.set_sqlite_path(tmp_path / "monitor.db") await monitor_db.init_db() transport = httpx.ASGITransport(app=app) async with httpx.AsyncClient(transport=transport, base_url="http://test") as http_client: yield http_client await monitor_db.dispose_engine() class TestParseTargetInput: def test_full_url_splits_id_from_token(self): """The id is the stable key; the token is a refreshable credential.""" parsed = parse_target_input(CREATOR_URL, "creator") assert parsed["external_id"] == "5f58bd990000000001003753" assert parsed["xsec_token"].startswith("ABYVg1evluJZZzpMX") assert parsed["xsec_source"] == "pc_search" def test_bare_id_is_accepted(self): parsed = parse_target_input("5f58bd990000000001003753", "creator") assert parsed["external_id"] == "5f58bd990000000001003753" assert parsed["xsec_token"] == "" def test_note_url_without_token_still_parses(self): parsed = parse_target_input( "https://www.xiaohongshu.com/explore/6aa3d827000000002802c5c8", "note" ) assert parsed["external_id"] == "6aa3d827000000002802c5c8" assert parsed["xsec_token"] == "" def test_creator_url_rejected_in_note_mode(self): with pytest.raises(TargetParseError): parse_target_input(CREATOR_URL, "note") def test_garbage_is_rejected(self): with pytest.raises(TargetParseError): parse_target_input("not a url at all !!", "creator") # --- 抖音 ------------------------------------------------------------- # 链接形态由平台决定,所以每一个都要显式带上 "dy"。 def test_douyin_creator_url(self): parsed = parse_target_input( "https://www.douyin.com/user/MS4wLjABAAAATJPY7LAlaa5X-c8uNdWkvz0jUGgpw4eeXIwu_8BhvqE" "?from_tab_name=main", "creator", "dy", ) assert ( parsed["external_id"] == "MS4wLjABAAAATJPY7LAlaa5X-c8uNdWkvz0jUGgpw4eeXIwu_8BhvqE" ) def test_douyin_video_url(self): parsed = parse_target_input( "https://www.douyin.com/video/7525082444551310602", "note", "dy" ) assert parsed["external_id"] == "7525082444551310602" def test_douyin_modal_id_url(self): """在别人主页或搜索结果里点开视频,拿到的就是带 modal_id 的链接。""" parsed = parse_target_input( "https://www.douyin.com/root/search/python?aid=b733a3b0&modal_id=7471165520058862848", "note", "dy", ) assert parsed["external_id"] == "7471165520058862848" def test_douyin_bare_sec_uid_is_accepted(self): sec_uid = "MS4wLjABAAAATJPY7LAlaa5X-c8uNdWkvz0jUGgpw4eeXIwu_8BhvqE" parsed = parse_target_input(sec_uid, "creator", "dy") assert parsed["external_id"] == sec_uid def test_douyin_bare_sec_uid_beyond_the_xhs_length_cap(self): """裸 id 的长度上限必须按平台分开。 小红书那条规则封顶 64 字符,而 sec_user_id 长过 64 是常态(实测样本 55, 但字段本身是变长的)。共用一条规则的话,长一点的 sec_uid 会被直接拒掉 —— 对用户来说就是「粘贴了一个完全正确的链接却报无法识别」。 """ sec_uid = "MS4wLjABAAAA" + "aB3dEf6hIj9lMn2pQr5tUv8xYz1" * 3 assert len(sec_uid) > 64 parsed = parse_target_input(sec_uid, "creator", "dy") assert parsed["external_id"] == sec_uid # 同一条 id 拿小红书规则来解析会被拒 —— 这正是两条规则必须分开的原因。 with pytest.raises(TargetParseError): parse_target_input(sec_uid, "creator", "xhs") def test_douyin_bare_video_id(self): parsed = parse_target_input("7525082444551310602", "note", "dy") assert parsed["external_id"] == "7525082444551310602" # 抖音不需要 xsec_token —— 和小红书不同,裸链接就能用。 assert parsed["xsec_token"] == "" def test_douyin_short_link_is_rejected_with_a_reason(self): """短链要联网跳一次才知道指向谁。明确拒绝好过存一个永远抓不到东西的目标。""" with pytest.raises(TargetParseError) as excinfo: parse_target_input("https://v.douyin.com/drIPtQ_WPWY/", "note", "dy") assert "短链" in str(excinfo.value) def test_a_douyin_link_is_not_parsed_with_xhs_rules(self): with pytest.raises(TargetParseError): parse_target_input( "https://www.douyin.com/video/7525082444551310602", "note", "xhs" ) def test_an_xhs_link_is_not_parsed_for_douyin(self): with pytest.raises(TargetParseError): parse_target_input(NOTE_URL, "note", "dy") def test_a_platform_without_an_adapter_is_rejected(self): with pytest.raises(TargetParseError): parse_target_input("whatever", "creator", "bili") class TestTargetReplacement: @pytest.mark.asyncio async def test_replacing_targets_uses_the_tasks_own_platform(self, client): """改目标必须按任务**自己**的平台解析。 ``update_task`` 原先漏传了 platform,解析回落到默认的小红书。只有小红书时 行为恰好正确,接上抖音就会拿小红书的正则去解析抖音链接 —— 建任务时对、 改任务时错,是最难注意到的那种不一致。 """ sec_uid = "MS4wLjABAAAATJPY7LAlaa5X-c8uNdWkvz0jUGgpw4eeXIwu_8BhvqE" created = await client.post( "/api/monitor/tasks", json={"name": "抖音", "mode": "creator", "platform": "dy", "targets": [sec_uid]}, ) assert created.status_code == 201 task_id = created.json()["id"] updated = await client.patch( f"/api/monitor/tasks/{task_id}", json={"targets": [f"https://www.douyin.com/user/{sec_uid}"]}, ) assert updated.status_code == 200 tasks = ( await client.get("/api/monitor/tasks", params={"platform": "dy"}) ).json()["tasks"] target = tasks[0]["targets"][0] assert target["external_id"] == sec_uid assert target["raw_value"].startswith("https://www.douyin.com/user/") class TestTaskCrud: @pytest.mark.asyncio async def test_create_and_list_task(self, client): response = await client.post( "/api/monitor/tasks", json={ "name": "网文作者监控", "mode": "creator", "interval_minutes": 120, "targets": [CREATOR_URL, "5f58bd990000000001003754"], }, ) assert response.status_code == 201 task_id = response.json()["id"] listing = await client.get("/api/monitor/tasks") assert listing.status_code == 200 tasks = listing.json()["tasks"] assert len(tasks) == 1 assert tasks[0]["id"] == task_id assert tasks[0]["target_count"] == 2 # next_run_at is persisted so the schedule survives a restart. assert tasks[0]["next_run_at"] is not None @pytest.mark.asyncio async def test_duplicate_targets_are_deduplicated(self, client): response = await client.post( "/api/monitor/tasks", json={ "name": "dedup", "mode": "creator", "targets": [CREATOR_URL, CREATOR_URL], }, ) assert response.status_code == 201 listing = await client.get("/api/monitor/tasks") assert listing.json()["tasks"][0]["target_count"] == 1 @pytest.mark.asyncio async def test_invalid_target_returns_400(self, client): response = await client.post( "/api/monitor/tasks", json={"name": "bad", "mode": "creator", "targets": ["!!! nonsense !!!"]}, ) assert response.status_code == 400 @pytest.mark.asyncio async def test_interval_floor_is_enforced(self, client): """A tight poll loop is the pattern that triggers platform rate limits.""" response = await client.post( "/api/monitor/tasks", json={"name": "too fast", "mode": "creator", "interval_minutes": 1, "targets": [CREATOR_URL]}, ) assert response.status_code == 422 @pytest.mark.asyncio async def test_update_and_delete(self, client): created = await client.post( "/api/monitor/tasks", json={"name": "t", "mode": "note", "targets": [NOTE_URL]}, ) task_id = created.json()["id"] patched = await client.patch(f"/api/monitor/tasks/{task_id}", json={"enabled": False}) assert patched.status_code == 200 listing = await client.get("/api/monitor/tasks") assert listing.json()["tasks"][0]["enabled"] is False deleted = await client.delete(f"/api/monitor/tasks/{task_id}") assert deleted.status_code == 200 assert (await client.get("/api/monitor/tasks")).json()["tasks"] == [] @pytest.mark.asyncio async def test_run_now_on_missing_task_is_404(self, client): response = await client.post("/api/monitor/tasks/9999/run") assert response.status_code == 404 @pytest.mark.asyncio async def test_run_history_starts_empty(self, client): created = await client.post( "/api/monitor/tasks", json={"name": "t", "mode": "creator", "targets": [CREATOR_URL]}, ) task_id = created.json()["id"] runs = await client.get(f"/api/monitor/tasks/{task_id}/runs") assert runs.status_code == 200 assert runs.json()["runs"] == [] class TestCookieEndpoints: @pytest.mark.asyncio async def test_cookie_value_is_never_returned(self, client): """The GET must expose health only, never the credential.""" secret = "web_session=SUPERSECRETVALUE; a1=abc123" saved = await client.post("/api/monitor/cookie", json={"cookie": secret}) assert saved.status_code == 200 status_response = await client.get("/api/monitor/cookie") assert status_response.status_code == 200 body = status_response.json() assert body["present"] is True assert body["length"] == len(secret) assert "SUPERSECRETVALUE" not in status_response.text @pytest.mark.asyncio async def test_cookie_initially_absent_and_clearable(self, client): assert (await client.get("/api/monitor/cookie")).json()["present"] is False await client.post("/api/monitor/cookie", json={"cookie": "web_session=x"}) assert (await client.get("/api/monitor/cookie")).json()["present"] is True await client.delete("/api/monitor/cookie") assert (await client.get("/api/monitor/cookie")).json()["present"] is False class TestDashboardQueries: @pytest.mark.asyncio async def test_empty_dashboard_shapes(self, client): assert (await client.get("/api/monitor/notes")).json()["notes"] == [] assert (await client.get("/api/monitor/comments")).json()["comments"] == [] assert (await client.get("/api/monitor/events")).json()["events"] == [] overview = (await client.get("/api/monitor/overview")).json() assert overview["tasks"] == 0 assert overview["notes"] == 0