"""关键词搜索离线测试:纯 parser + search_keyword(去重/翻页/截断/SearchError)。""" from __future__ import annotations import json from pathlib import Path import pytest from creation_knowledge.config import PgConfig, Settings from creation_knowledge.integrations.crawler import RateLimiter from creation_knowledge.integrations.search import ( SearchError, parse_search_response, search_keyword, ) FIX = Path(__file__).parent / "fixtures" def _settings() -> Settings: return Settings( pg=PgConfig(host="h", port=5432, user="u", password="p", database="d"), crawler_base_url="http://crawler.test", crawler_key="", crawler_timeout=30, video_model="m", gemini_api_key="", openrouter_base_url="b", openrouter_api_key="k", llm_model="m", max_cards=12, frames_dir="f", douyin_ratio="540p", data_dir="") def _no_wait() -> RateLimiter: return RateLimiter(min_interval_seconds=0.0) class _Resp: def __init__(self, payload): self._p = payload def raise_for_status(self): return None def json(self): return self._p class _FakeClient: """按调用次序返回预置 payload;记录每次 post 的 body。""" def __init__(self, pages): self.pages = list(pages); self.calls = []; self.closed = False def post(self, url, json=None, headers=None, timeout=None): self.calls.append(json) return _Resp(self.pages.pop(0)) def close(self): self.closed = True def test_parse_from_fixture(): resp = json.loads((FIX / "xhs_search_分镜脚本.json").read_text("utf-8")) ids, has_more, cursor = parse_search_response(resp) assert ids == ["6a2bce890000000006034be4", "68282529000000002100e28c"] assert has_more is True and cursor == "2" def test_parse_business_error_raises(): with pytest.raises(SearchError): parse_search_response({"code": 10000, "msg": "未知错误", "data": None}) def test_search_dedup_and_limit(): # page1 含重复 id,page2 提供更多;limit=3 → 截断且去重 page1 = {"code": 0, "data": {"has_more": True, "next_cursor": "2", "data": [{"id": "a"}, {"id": "a"}, {"id": "b"}]}} page2 = {"code": 0, "data": {"has_more": True, "next_cursor": "3", "data": [{"id": "b"}, {"id": "c"}, {"id": "d"}]}} client = _FakeClient([page1, page2]) out = search_keyword("分镜", settings=_settings(), http_client=client, rate_limiter=_no_wait(), limit=3) assert out == ["a", "b", "c"] def test_search_stops_on_no_more(): page1 = {"code": 0, "data": {"has_more": False, "next_cursor": "", "data": [{"id": "a"}, {"id": "b"}]}} client = _FakeClient([page1]) out = search_keyword("分镜", settings=_settings(), http_client=client, rate_limiter=_no_wait(), limit=10) assert out == ["a", "b"] # has_more=False → 不再翻页,不报错 assert client.calls[0]["keyword"] == "分镜" def test_search_business_error_raises(): client = _FakeClient([{"code": 10000, "msg": "未知错误", "data": None}]) with pytest.raises(SearchError): search_keyword("x", settings=_settings(), http_client=client, rate_limiter=_no_wait()) def test_search_unsupported_platform(): with pytest.raises(SearchError): search_keyword("x", platform="bilibili", settings=_settings(), rate_limiter=_no_wait())