test_search.py 7.3 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170
  1. """关键词搜索离线测试:纯 parser + search_keyword(去重/翻页/截断/SearchError)。"""
  2. from __future__ import annotations
  3. import json
  4. from pathlib import Path
  5. import pytest
  6. from core.config import PgConfig, Settings
  7. from acquisition.crawler import RateLimiter
  8. from acquisition.search import (
  9. SearchError, parse_search_response, parse_weixin, parse_xiaohongshu, search_keyword,
  10. )
  11. FIX = Path(__file__).parent / "fixtures"
  12. def _settings() -> Settings:
  13. return Settings(
  14. pg=PgConfig(host="h", port=5432, user="u", password="p", database="d"),
  15. crawler_base_url="http://crawler.test", crawler_key="", crawler_timeout=30,
  16. video_model="m", gemini_api_key="",
  17. openrouter_base_url="b", openrouter_api_key="k",
  18. llm_model="m", max_cards=12, frames_dir="f", douyin_ratio="540p", data_dir="")
  19. def _no_wait() -> RateLimiter:
  20. return RateLimiter(min_interval_seconds=0.0)
  21. class _Resp:
  22. def __init__(self, payload): self._p = payload
  23. def raise_for_status(self): return None
  24. def json(self): return self._p
  25. class _FakeClient:
  26. """按调用次序返回预置 payload;记录每次 post 的 body。"""
  27. def __init__(self, pages): self.pages = list(pages); self.calls = []; self.closed = False
  28. def post(self, url, json=None, headers=None, timeout=None):
  29. self.calls.append(json)
  30. return _Resp(self.pages.pop(0))
  31. def close(self): self.closed = True
  32. def test_parse_from_fixture():
  33. resp = json.loads((FIX / "xhs_search_分镜脚本.json").read_text("utf-8"))
  34. ids, has_more, cursor = parse_search_response(resp)
  35. assert ids == ["6a2bce890000000006034be4", "68282529000000002100e28c"]
  36. assert has_more is True and cursor == "2"
  37. def test_parse_business_error_raises():
  38. with pytest.raises(SearchError):
  39. parse_search_response({"code": 10000, "msg": "未知错误", "data": None})
  40. def test_search_dedup_and_limit():
  41. # page1 含重复 id,page2 提供更多;limit=3 → 截断且去重
  42. page1 = {"code": 0, "data": {"has_more": True, "next_cursor": "2",
  43. "data": [{"id": "a"}, {"id": "a"}, {"id": "b"}]}}
  44. page2 = {"code": 0, "data": {"has_more": True, "next_cursor": "3",
  45. "data": [{"id": "b"}, {"id": "c"}, {"id": "d"}]}}
  46. client = _FakeClient([page1, page2])
  47. out = search_keyword("分镜", settings=_settings(), http_client=client,
  48. rate_limiter=_no_wait(), limit=3)
  49. assert out == ["a", "b", "c"]
  50. def test_search_stops_on_no_more():
  51. page1 = {"code": 0, "data": {"has_more": False, "next_cursor": "",
  52. "data": [{"id": "a"}, {"id": "b"}]}}
  53. client = _FakeClient([page1])
  54. out = search_keyword("分镜", settings=_settings(), http_client=client,
  55. rate_limiter=_no_wait(), limit=10)
  56. assert out == ["a", "b"] # has_more=False → 不再翻页,不报错
  57. assert client.calls[0]["keyword"] == "分镜"
  58. def test_search_business_error_raises():
  59. client = _FakeClient([{"code": 10000, "msg": "未知错误", "data": None}])
  60. with pytest.raises(SearchError):
  61. search_keyword("x", settings=_settings(), http_client=client,
  62. rate_limiter=_no_wait())
  63. def test_search_unsupported_platform():
  64. with pytest.raises(SearchError):
  65. search_keyword("x", platform="bilibili", settings=_settings(),
  66. rate_limiter=_no_wait())
  67. def test_parse_douyin_uses_aweme_id():
  68. resp = {"code": 0, "data": {"has_more": True, "next_cursor": "10",
  69. "data": [{"aweme_id": "7618138722512424246"}, {"aweme_id": "7600000000000000000"}]}}
  70. ids, has_more, cursor = parse_search_response(resp, platform="douyin")
  71. assert ids == ["7618138722512424246", "7600000000000000000"]
  72. assert has_more is True and cursor == "10"
  73. def test_douyin_body_has_account_id_and_params():
  74. page = {"code": 0, "data": {"has_more": False, "next_cursor": "",
  75. "data": [{"aweme_id": "a1"}, {"aweme_id": "a2"}]}}
  76. client = _FakeClient([page])
  77. out = search_keyword("科普", platform="douyin", content_type="视频",
  78. settings=_settings(), http_client=client, rate_limiter=_no_wait(), limit=10)
  79. assert out == ["a1", "a2"] # 用 aweme_id 解析
  80. body = client.calls[0]
  81. assert body["account_id"] == "7450041106378522636" # piaoquantv 抖音账号(settings.douyin_account_id)
  82. assert body["cookie_batch"] == "default" # piaoquantv 抖音必带
  83. assert body["sort_type"] == "综合排序" # 平台默认(非小红书的"综合")
  84. assert body["publish_time"] == "不限"
  85. assert body["cursor"] == "0" # 抖音起始 cursor
  86. def test_parse_weixin_cleans_title_and_keeps_url_cover():
  87. resp = {"code": 0, "data": {"data": [
  88. {"title": "复旦教授<em class='highlight'>历史科普</em>", "url": "http://mp.weixin.qq.com/s?x=1",
  89. "cover_url": "https://mmbiz.qpic.cn/a.jpg", "nick_name": "某号", "time": "1天前"},
  90. {"title": "无链接的应被跳过"}, # 无 url → 跳过
  91. ]}}
  92. out = parse_weixin(resp, limit=5)
  93. assert len(out) == 1
  94. assert out[0]["title"] == "复旦教授历史科普" # <em> 标记被清掉
  95. assert out[0]["url"].startswith("http://mp.weixin.qq.com")
  96. assert out[0]["cover_url"].endswith("a.jpg") and out[0]["nick_name"] == "某号"
  97. def test_parse_weixin_business_error():
  98. with pytest.raises(SearchError):
  99. parse_weixin({"code": 10000, "msg": "x"})
  100. def test_parse_xiaohongshu_takes_cover_title_from_note_card():
  101. resp = {"code": 0, "data": {"data": [
  102. {"id": "abc123", "note_card": {
  103. "type": "video", "display_title": "社会运行规则",
  104. "image_list": [{"image_url": "https://ci.xiaohongshu.com/cover.jpg"}],
  105. "user": {"nickname": "某号"}}},
  106. {"id": "noimg", "note_card": {"display_title": "无封面应跳过", "image_list": []}},
  107. ]}}
  108. out = parse_xiaohongshu(resp, limit=5)
  109. assert len(out) == 1 # 无封面的被跳过
  110. assert out[0]["id"] == "abc123" and out[0]["title"] == "社会运行规则"
  111. assert out[0]["cover_url"].endswith("cover.jpg") and out[0]["nick_name"] == "某号"
  112. assert out[0]["url"] == "https://www.xiaohongshu.com/explore/abc123"
  113. def test_parse_xiaohongshu_title_falls_back_to_desc():
  114. resp = {"code": 0, "data": {"data": [
  115. {"id": "x1", "note_card": {"display_title": "", "desc": "那些赤裸裸的社会真相!\n第二行",
  116. "image_list": [{"image_url": "u.jpg"}]}},
  117. ]}}
  118. out = parse_xiaohongshu(resp, limit=5)
  119. assert out[0]["title"] == "那些赤裸裸的社会真相!" # display_title 空 → 取 desc 首行
  120. def test_parse_xiaohongshu_business_error():
  121. with pytest.raises(SearchError):
  122. parse_xiaohongshu({"code": 10000, "msg": "x"})
  123. def test_xiaohongshu_body_no_account_id():
  124. page = {"code": 0, "data": {"has_more": False, "next_cursor": "", "data": [{"id": "x1"}]}}
  125. client = _FakeClient([page])
  126. search_keyword("科普", platform="xiaohongshu", settings=_settings(),
  127. http_client=client, rate_limiter=_no_wait())
  128. body = client.calls[0]
  129. assert "account_id" not in body # 小红书不带
  130. assert body["sort_type"] == "综合" and body["cursor"] == ""