test_crawler.py 7.5 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217
  1. """M2 离线测试:用 5 个真实 capture 验证 parse_detail_response 字段映射。"""
  2. from __future__ import annotations
  3. import json
  4. from pathlib import Path
  5. import pytest
  6. from acquisition.crawler import (
  7. PLATFORMS,
  8. RateLimiter,
  9. detect_platform_and_id,
  10. fetch_post_detail,
  11. parse_content_id,
  12. parse_detail_response,
  13. )
  14. from core.config import PgConfig, Settings
  15. FIXTURES = Path(__file__).parent / "fixtures"
  16. # content_id -> (期望作者, 标题里应出现的子串)
  17. EXPECTED = {
  18. "67e4bdf50000000006028a59": ("海狸教自媒体运营", "短视频脚本"),
  19. "698481e1000000000a02a7c1": ("Irvin是个编剧(接稿版)", "叙事结构"),
  20. "67e2e39b0000000003028ff0": ("拾意", "剧本创作"),
  21. "699308fa0000000016009697": ("方圆的增长飞轮", "选题"),
  22. "680659e8000000001a007a11": ("故事设计原理拆解学习"[:2], "故事设计"),
  23. }
  24. def _load(content_id: str) -> dict:
  25. return json.loads((FIXTURES / f"xhs_case_{content_id}.json").read_text("utf-8"))
  26. def _settings() -> Settings:
  27. return Settings(
  28. pg=PgConfig(host="h", port=5432, user="u", password="p", database="d"),
  29. aiddit_crawler_base_url="http://crawler.test",
  30. crawler_timeout=30,
  31. openrouter_timeout_seconds=90,
  32. openrouter_model="m",
  33. openrouter_base_url="b",
  34. openrouter_api_key="k",
  35. llm_model="m",
  36. max_cards=12,
  37. frames_dir="f",
  38. douyin_ratio="540p",
  39. data_dir="",
  40. )
  41. class _Resp:
  42. def __init__(self, payload):
  43. self._p = payload
  44. def raise_for_status(self):
  45. return None
  46. def json(self):
  47. return self._p
  48. class _FakeClient:
  49. def __init__(self, payload):
  50. self.payload = payload
  51. self.calls = []
  52. self.closed = False
  53. def post(self, url, json=None, headers=None, timeout=None):
  54. self.calls.append({"url": url, "body": json})
  55. return _Resp(self.payload)
  56. def close(self):
  57. self.closed = True
  58. class _RecordingLimiter:
  59. def __init__(self):
  60. self.buckets = []
  61. def wait(self, bucket: str):
  62. self.buckets.append(bucket)
  63. @pytest.mark.parametrize("content_id", list(EXPECTED))
  64. def test_parse_detail_fields(content_id: str):
  65. # 重抓的 xhs_case_*.json 即原始响应 {code,msg,data},无 response 包裹
  66. response = _load(content_id)
  67. post = parse_detail_response(response, fallback_content_id=content_id)
  68. assert post.id == f"xhs_{content_id}"
  69. assert post.platform == "xiaohongshu"
  70. assert post.content_id == content_id
  71. assert content_id in post.url
  72. assert post.title, "title 不应为空"
  73. _, title_sub = EXPECTED[content_id]
  74. assert title_sub in post.title
  75. assert post.raw.get("code") == 0
  76. def test_authors():
  77. for cid, (author, _) in EXPECTED.items():
  78. post = parse_detail_response(_load(cid), fallback_content_id=cid)
  79. if cid in ("67e4bdf50000000006028a59", "698481e1000000000a02a7c1",
  80. "67e2e39b0000000003028ff0", "699308fa0000000016009697"):
  81. assert post.author_name == author
  82. def test_shiyi_posts_text_sparse():
  83. """拾意两条正文几乎只有话题串——M3 多模态提取存在的理由。"""
  84. for cid in ("67e2e39b0000000003028ff0", "680659e8000000001a007a11"):
  85. post = parse_detail_response(_load(cid), fallback_content_id=cid)
  86. stripped = post.body_text.replace("#", "").replace("话题", "").strip()
  87. assert len(stripped) < 40, f"{cid} body_text 应很稀疏: {post.body_text!r}"
  88. def test_parse_content_id_from_url():
  89. url = ("https://www.xiaohongshu.com/explore/67e4bdf50000000006028a59"
  90. "?xsec_token=ABC=&xsec_source=pc_like")
  91. assert parse_content_id(url) == "67e4bdf50000000006028a59"
  92. assert parse_content_id("67e4bdf50000000006028a59") == "67e4bdf50000000006028a59"
  93. @pytest.mark.parametrize("url,platform,cid", [
  94. # 长链
  95. ("https://www.xiaohongshu.com/explore/67e4bdf50000000006028a59?xsec_token=A",
  96. "xiaohongshu", "67e4bdf50000000006028a59"),
  97. ("https://www.xiaohongshu.com/discovery/item/680659e8000000001a007a11",
  98. "xiaohongshu", "680659e8000000001a007a11"),
  99. ("https://www.douyin.com/video/7612631899479648866", "douyin", "7612631899479648866"),
  100. ("https://www.douyin.com/search/x?modal_id=7612631899479648866&type=general",
  101. "douyin", "7612631899479648866"),
  102. ("https://www.gifshow.com/fw/photo/3xepqgwddgbickc", "kuaishou", "3xepqgwddgbickc"),
  103. ("https://www.kuaishou.com/short-video/3xepqgwddgbickc", "kuaishou", "3xepqgwddgbickc"),
  104. ("https://www.bilibili.com/video/BV1qd4y1x76w/?spm_id_from=333", "bilibili", "BV1qd4y1x76w"),
  105. # 裸 id 兜底
  106. ("67e4bdf50000000006028a59", "xiaohongshu", "67e4bdf50000000006028a59"),
  107. ("7612631899479648866", "douyin", "7612631899479648866"),
  108. ("BV1qd4y1x76w", "bilibili", "BV1qd4y1x76w"),
  109. ])
  110. def test_detect_platform_and_id(url, platform, cid):
  111. assert detect_platform_and_id(url) == (platform, cid)
  112. def test_parse_detail_platform_prefix():
  113. """同一 parse 函数 + platform 参数 → 平台名与 id 前缀正确。"""
  114. response = _load("67e4bdf50000000006028a59")
  115. for platform, expected_prefix in [(p, c["prefix"]) for p, c in PLATFORMS.items()]:
  116. post = parse_detail_response(response, platform=platform,
  117. fallback_content_id="67e4bdf50000000006028a59")
  118. assert post.platform == platform
  119. assert post.id == f"{expected_prefix}_67e4bdf50000000006028a59"
  120. def test_fetch_douyin_detail_piaoquantv_body_and_provider():
  121. payload = {"code": 0, "data": {"data": {
  122. "channel_content_id": "761",
  123. "content_link": "https://douyin.test/video/761",
  124. "content_type": "video",
  125. "title": "视频",
  126. "video_url_list": [{"video_url": "https://video.test/761.mp4"}],
  127. }}}
  128. client = _FakeClient(payload)
  129. limiter = _RecordingLimiter()
  130. post = fetch_post_detail(
  131. "761",
  132. platform="douyin",
  133. provider="piaoquantv",
  134. settings=_settings(),
  135. http_client=client,
  136. rate_limiter=limiter,
  137. )
  138. assert post.provider == "piaoquantv"
  139. assert post.raw["detail_provider"] == "piaoquantv"
  140. assert client.calls[0]["url"].startswith("http://crawapi.piaoquantv.com/")
  141. assert client.calls[0]["body"]["content_id"] == "761"
  142. assert client.calls[0]["body"]["account_id"] == "7450041106378522636"
  143. assert client.calls[0]["body"]["cookie_batch"] == "default"
  144. assert limiter.buckets == ["douyin"]
  145. def test_fetch_douyin_detail_aiddit_body_and_provider():
  146. payload = {"code": 0, "data": {"data": {
  147. "channel_content_id": "762",
  148. "content_link": "https://douyin.test/video/762",
  149. "content_type": "video",
  150. "title": "视频",
  151. "video_url_list": [{"video_url": "https://video.test/762.mp4"}],
  152. }}}
  153. client = _FakeClient(payload)
  154. post = fetch_post_detail(
  155. "762",
  156. platform="douyin",
  157. provider="aiddit",
  158. settings=_settings(),
  159. http_client=client,
  160. rate_limiter=_RecordingLimiter(),
  161. )
  162. assert post.provider == "aiddit"
  163. assert client.calls[0]["url"].startswith("http://crawler.test/")
  164. assert client.calls[0]["body"] == {"content_id": "762"}
  165. def test_fetch_douyin_detail_rejects_unknown_provider():
  166. with pytest.raises(RuntimeError):
  167. fetch_post_detail(
  168. "762",
  169. platform="douyin",
  170. provider="other",
  171. settings=_settings(),
  172. http_client=_FakeClient({}),
  173. rate_limiter=RateLimiter(min_interval_seconds=0.0),
  174. )