"""关键词搜索离线测试:纯 parser + search_keyword(去重/翻页/截断/SearchError)。"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
from core.config import PgConfig, Settings
from acquisition.crawler import RateLimiter
from acquisition.search import (
SearchError, parse_search_response, parse_weixin, parse_xiaohongshu, search_keyword,
)
FIX = Path(__file__).parent / "fixtures"
def _settings() -> Settings:
return Settings(
pg=PgConfig(host="h", port=5432, user="u", password="p", database="d"),
crawler_base_url="http://crawler.test", crawler_key="", crawler_timeout=30,
video_model="m", gemini_api_key="",
openrouter_base_url="b", openrouter_api_key="k",
llm_model="m", max_cards=12, frames_dir="f", douyin_ratio="540p", data_dir="")
def _no_wait() -> RateLimiter:
return RateLimiter(min_interval_seconds=0.0)
class _Resp:
def __init__(self, payload): self._p = payload
def raise_for_status(self): return None
def json(self): return self._p
class _FakeClient:
"""按调用次序返回预置 payload;记录每次 post 的 body。"""
def __init__(self, pages): self.pages = list(pages); self.calls = []; self.closed = False
def post(self, url, json=None, headers=None, timeout=None):
self.calls.append(json)
return _Resp(self.pages.pop(0))
def close(self): self.closed = True
def test_parse_from_fixture():
resp = json.loads((FIX / "xhs_search_分镜脚本.json").read_text("utf-8"))
ids, has_more, cursor = parse_search_response(resp)
assert ids == ["6a2bce890000000006034be4", "68282529000000002100e28c"]
assert has_more is True and cursor == "2"
def test_parse_business_error_raises():
with pytest.raises(SearchError):
parse_search_response({"code": 10000, "msg": "未知错误", "data": None})
def test_search_dedup_and_limit():
# page1 含重复 id,page2 提供更多;limit=3 → 截断且去重
page1 = {"code": 0, "data": {"has_more": True, "next_cursor": "2",
"data": [{"id": "a"}, {"id": "a"}, {"id": "b"}]}}
page2 = {"code": 0, "data": {"has_more": True, "next_cursor": "3",
"data": [{"id": "b"}, {"id": "c"}, {"id": "d"}]}}
client = _FakeClient([page1, page2])
out = search_keyword("分镜", settings=_settings(), http_client=client,
rate_limiter=_no_wait(), limit=3)
assert out == ["a", "b", "c"]
def test_search_stops_on_no_more():
page1 = {"code": 0, "data": {"has_more": False, "next_cursor": "",
"data": [{"id": "a"}, {"id": "b"}]}}
client = _FakeClient([page1])
out = search_keyword("分镜", settings=_settings(), http_client=client,
rate_limiter=_no_wait(), limit=10)
assert out == ["a", "b"] # has_more=False → 不再翻页,不报错
assert client.calls[0]["keyword"] == "分镜"
def test_search_business_error_raises():
client = _FakeClient([{"code": 10000, "msg": "未知错误", "data": None}])
with pytest.raises(SearchError):
search_keyword("x", settings=_settings(), http_client=client,
rate_limiter=_no_wait())
def test_search_unsupported_platform():
with pytest.raises(SearchError):
search_keyword("x", platform="bilibili", settings=_settings(),
rate_limiter=_no_wait())
def test_parse_douyin_uses_aweme_id():
resp = {"code": 0, "data": {"has_more": True, "next_cursor": "10",
"data": [{"aweme_id": "7618138722512424246"}, {"aweme_id": "7600000000000000000"}]}}
ids, has_more, cursor = parse_search_response(resp, platform="douyin")
assert ids == ["7618138722512424246", "7600000000000000000"]
assert has_more is True and cursor == "10"
def test_douyin_body_has_account_id_and_params():
page = {"code": 0, "data": {"has_more": False, "next_cursor": "",
"data": [{"aweme_id": "a1"}, {"aweme_id": "a2"}]}}
client = _FakeClient([page])
out = search_keyword("科普", platform="douyin", content_type="视频",
settings=_settings(), http_client=client, rate_limiter=_no_wait(), limit=10)
assert out == ["a1", "a2"] # 用 aweme_id 解析
body = client.calls[0]
assert body["account_id"] == "7450041106378522636" # piaoquantv 抖音账号(settings.douyin_account_id)
assert body["cookie_batch"] == "default" # piaoquantv 抖音必带
assert body["sort_type"] == "综合排序" # 平台默认(非小红书的"综合")
assert body["publish_time"] == "不限"
assert body["cursor"] == "0" # 抖音起始 cursor
def test_parse_weixin_cleans_title_and_keeps_url_cover():
resp = {"code": 0, "data": {"data": [
{"title": "复旦教授历史科普", "url": "http://mp.weixin.qq.com/s?x=1",
"cover_url": "https://mmbiz.qpic.cn/a.jpg", "nick_name": "某号", "time": "1天前"},
{"title": "无链接的应被跳过"}, # 无 url → 跳过
]}}
out = parse_weixin(resp, limit=5)
assert len(out) == 1
assert out[0]["title"] == "复旦教授历史科普" # 标记被清掉
assert out[0]["url"].startswith("http://mp.weixin.qq.com")
assert out[0]["cover_url"].endswith("a.jpg") and out[0]["nick_name"] == "某号"
def test_parse_weixin_business_error():
with pytest.raises(SearchError):
parse_weixin({"code": 10000, "msg": "x"})
def test_parse_xiaohongshu_takes_cover_title_from_note_card():
resp = {"code": 0, "data": {"data": [
{"id": "abc123", "note_card": {
"type": "video", "display_title": "社会运行规则",
"image_list": [{"image_url": "https://ci.xiaohongshu.com/cover.jpg"}],
"user": {"nickname": "某号"}}},
{"id": "noimg", "note_card": {"display_title": "无封面应跳过", "image_list": []}},
]}}
out = parse_xiaohongshu(resp, limit=5)
assert len(out) == 1 # 无封面的被跳过
assert out[0]["id"] == "abc123" and out[0]["title"] == "社会运行规则"
assert out[0]["cover_url"].endswith("cover.jpg") and out[0]["nick_name"] == "某号"
assert out[0]["url"] == "https://www.xiaohongshu.com/explore/abc123"
def test_parse_xiaohongshu_title_falls_back_to_desc():
resp = {"code": 0, "data": {"data": [
{"id": "x1", "note_card": {"display_title": "", "desc": "那些赤裸裸的社会真相!\n第二行",
"image_list": [{"image_url": "u.jpg"}]}},
]}}
out = parse_xiaohongshu(resp, limit=5)
assert out[0]["title"] == "那些赤裸裸的社会真相!" # display_title 空 → 取 desc 首行
def test_parse_xiaohongshu_business_error():
with pytest.raises(SearchError):
parse_xiaohongshu({"code": 10000, "msg": "x"})
def test_xiaohongshu_body_no_account_id():
page = {"code": 0, "data": {"has_more": False, "next_cursor": "", "data": [{"id": "x1"}]}}
client = _FakeClient([page])
search_keyword("科普", platform="xiaohongshu", settings=_settings(),
http_client=client, rate_limiter=_no_wait())
body = client.calls[0]
assert "account_id" not in body # 小红书不带
assert body["sort_type"] == "综合" and body["cursor"] == ""