| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673 |
- """关键词搜索:query → content_id 列表(翻页 + 去重 + 截断)。
- 接口(实测 2026-06-26,host = settings.aiddit_crawler_base_url):
- 小红书 POST /crawler/xiao_hong_shu/keyword
- body{keyword, sort_type:"综合", publish_time:"", cursor:""}
- → {code:0, data:{has_more, next_cursor, data:[{id:<content_id>, note_card:{...}}]}}
- 抖音 POST /crawler/dou_yin/keyword
- body{keyword, content_type:"视频", sort_type:"综合排序", publish_time:"不限",
- cursor:"0", account_id:"771431222"} # ← 缺 account_id 会 code 10000
- → {code:0, data:{has_more, next_cursor, data:[{aweme_id:<content_id>, ...}]}}
- 各平台请求参数 / id 字段不同,集中在 PLATFORM_SEARCH 表里。
- 搜索只给预览;全文/全图/视频仍需 crawler.fetch_post_detail 逐 id 取详情(detail 只需 content_id)。
- parse_search_response 纯函数可离线测;search_keyword 负责 HTTP + 翻页。
- """
- from __future__ import annotations
- from dataclasses import dataclass, field
- import re
- from typing import Any, Optional
- from urllib.parse import urljoin
- import httpx
- from core.config import Settings
- from core.text_limits import TITLE_FALLBACK_MAX_CHARS, clip_text
- from acquisition.crawler import (
- CRAWLER_MAX_INTERVAL_SECONDS,
- CRAWLER_MIN_INTERVAL_SECONDS,
- RateLimiter,
- )
- # 每平台的搜索参数差异(path / 条目 id 字段 / 排序 / 发布时间 / 起始 cursor / account_id)
- PLATFORM_SEARCH = {
- "xiaohongshu": {
- "path": "/crawler/xiao_hong_shu/keyword", "id_key": "id",
- "sort_type": "综合", "publish_time": "", "cursor0": "", "account_id": None,
- },
- "douyin": {
- "path": "/crawler/dou_yin/keyword", "id_key": "aweme_id",
- "sort_type": "综合排序", "publish_time": "不限", "cursor0": "0",
- # account_id 改由 settings.piaoquantv_douyin_account_id 提供(抖音走 piaoquantv 后端);缺它搜索报错
- "account_id": None,
- },
- }
- class SearchError(RuntimeError):
- pass
- @dataclass(frozen=True)
- class SearchHit:
- content_id: str
- provider: str = ""
- raw: dict[str, Any] = field(default_factory=dict)
- request_payload: dict[str, Any] = field(default_factory=dict)
- @dataclass(frozen=True)
- class SearchPage:
- rows: list[Any]
- raw_count: int
- cursor: str = ""
- next_cursor: str = ""
- has_more: bool = False
- provider: str = ""
- request_payload: dict[str, Any] = field(default_factory=dict)
- raw_metadata: dict[str, Any] = field(default_factory=dict)
- DOUYIN_PROVIDERS = ("piaoquantv", "aiddit")
- def _douyin_cursor0(provider: str) -> str:
- return "" if provider == "aiddit" else "0"
- def _douyin_base_url(settings: Settings, provider: str) -> str:
- if provider == "piaoquantv":
- return settings.piaoquantv_douyin_base_url
- if provider == "aiddit":
- return settings.aiddit_crawler_base_url
- raise SearchError(f"unsupported douyin provider: {provider}")
- def _douyin_search_body(
- keyword: str,
- *,
- provider: str,
- cursor: str,
- settings: Settings,
- content_type: str | None,
- sort_type: str | None,
- publish_time: str | None,
- account_id: str | None,
- ) -> dict[str, Any]:
- if provider == "piaoquantv":
- body: dict[str, Any] = {
- "keyword": keyword,
- "sort_type": sort_type or "综合排序",
- "publish_time": publish_time or "不限",
- "cursor": cursor,
- }
- if content_type:
- body["content_type"] = content_type
- body["account_id"] = account_id or settings.piaoquantv_douyin_account_id
- body["cookie_batch"] = settings.piaoquantv_douyin_cookie_batch
- return body
- if provider == "aiddit":
- body = {
- "keyword": keyword,
- "cursor": cursor,
- "sort_type": sort_type or "最多点赞",
- }
- if content_type:
- body["content_type"] = content_type
- return body
- raise SearchError(f"unsupported douyin provider: {provider}")
- def _parse_search_items(response: Any, *, platform: str = "xiaohongshu") -> tuple[list[dict[str, Any]], int, bool, str]:
- if not isinstance(response, dict):
- raise SearchError("bad_response: not a dict")
- code = response.get("code")
- if code not in (0, "0"):
- raise SearchError(f"business_error: code={code} msg={response.get('msg')}")
- data = response.get("data") or {}
- items = data.get("data") or []
- if not isinstance(items, list):
- raise SearchError("bad_response: data.data is not a list")
- id_key = (PLATFORM_SEARCH.get(platform) or {}).get("id_key", "id")
- valid = [it for it in items if isinstance(it, dict) and it.get(id_key)]
- return valid, len(items), bool(data.get("has_more")), str(data.get("next_cursor") or "")
- def parse_search_response(response: Any, *, platform: str = "xiaohongshu") -> tuple[list[str], bool, str]:
- """信封 {code,msg,data:{has_more,next_cursor,data:[...]}} → (content_ids, has_more, next_cursor)。
- 条目 id 字段按平台取(小红书 id / 抖音 aweme_id)。code!=0 抛 SearchError。"""
- id_key = (PLATFORM_SEARCH.get(platform) or {}).get("id_key", "id")
- items, _, has_more, next_cursor = _parse_search_items(response, platform=platform)
- ids = [it[id_key] for it in items]
- return ids, has_more, next_cursor
- def search_douyin_page(
- keyword: str,
- *,
- provider: str = "piaoquantv",
- cursors: dict[str, str] | None = None,
- content_type: str | None = "视频",
- sort_type: Optional[str] = None,
- publish_time: Optional[str] = None,
- account_id: Optional[str] = None,
- limit: int = 10,
- settings: Optional[Settings] = None,
- http_client: Any = None,
- rate_limiter: Optional[RateLimiter] = None,
- env_file: str = ".env",
- ) -> SearchPage:
- """Fetch one Douyin page, using piaoquantv first and AIDDIT as fallback."""
- settings = settings or Settings.from_env(env_file)
- rate_limiter = rate_limiter or RateLimiter(
- min_interval_seconds=CRAWLER_MIN_INTERVAL_SECONDS,
- max_interval_seconds=CRAWLER_MAX_INTERVAL_SECONDS,
- )
- owns_client = http_client is None
- client = http_client or httpx.Client()
- cursors = dict(cursors or {p: _douyin_cursor0(p) for p in DOUYIN_PROVIDERS})
- provider_order = [provider] + [p for p in DOUYIN_PROVIDERS if p != provider]
- errors: list[str] = []
- try:
- for attempt_provider in provider_order:
- cursor = cursors.get(attempt_provider, _douyin_cursor0(attempt_provider))
- body = _douyin_search_body(
- keyword,
- provider=attempt_provider,
- cursor=cursor,
- settings=settings,
- content_type=content_type,
- sort_type=sort_type,
- publish_time=publish_time,
- account_id=account_id,
- )
- rate_limiter.wait("douyin")
- url = urljoin(_douyin_base_url(settings, attempt_provider), PLATFORM_SEARCH["douyin"]["path"])
- try:
- resp = client.post(
- url,
- json=body,
- headers={"Content-Type": "application/json"},
- timeout=settings.crawler_timeout,
- )
- resp.raise_for_status()
- data = resp.json()
- items, raw_count, has_more, next_cursor = _parse_search_items(data, platform="douyin")
- except httpx.HTTPError as exc:
- errors.append(f"{attempt_provider}: http_error: {exc}")
- continue
- except ValueError:
- errors.append(f"{attempt_provider}: bad_json")
- continue
- except SearchError as exc:
- errors.append(f"{attempt_provider}: {exc}")
- continue
- if not items:
- errors.append(f"{attempt_provider}: empty_aweme_id")
- continue
- id_key = PLATFORM_SEARCH["douyin"]["id_key"]
- hits = [
- SearchHit(
- content_id=str(item[id_key]),
- provider=attempt_provider,
- raw=item,
- request_payload=dict(body),
- )
- for item in items[:limit]
- ]
- next_cursors = dict(cursors)
- if has_more and next_cursor and next_cursor != cursor:
- next_cursors[attempt_provider] = next_cursor
- return SearchPage(
- rows=hits,
- raw_count=raw_count,
- cursor=cursor,
- next_cursor=next_cursor,
- has_more=has_more,
- provider=attempt_provider,
- request_payload=dict(body),
- raw_metadata={"cursors": next_cursors, "errors": errors},
- )
- finally:
- if owns_client:
- client.close()
- raise SearchError("; ".join(errors) or "douyin_search_failed")
- def search_douyin_hits(
- keyword: str,
- *,
- content_type: str | None = "视频",
- sort_type: Optional[str] = None,
- publish_time: Optional[str] = None,
- account_id: Optional[str] = None,
- limit: int = 5,
- settings: Optional[Settings] = None,
- http_client: Any = None,
- rate_limiter: Optional[RateLimiter] = None,
- env_file: str = ".env",
- max_pages: int = 10,
- ) -> list[SearchHit]:
- """Douyin search with piaoquantv primary and AIDDIT fallback per cursor."""
- settings = settings or Settings.from_env(env_file)
- rate_limiter = rate_limiter or RateLimiter(
- min_interval_seconds=CRAWLER_MIN_INTERVAL_SECONDS,
- max_interval_seconds=CRAWLER_MAX_INTERVAL_SECONDS,
- )
- owns_client = http_client is None
- client = http_client or httpx.Client()
- out: list[SearchHit] = []
- seen: set[str] = set()
- provider = "piaoquantv"
- cursors = {p: _douyin_cursor0(p) for p in DOUYIN_PROVIDERS}
- try:
- for _ in range(max_pages):
- page = search_douyin_page(
- keyword,
- provider=provider,
- cursors=cursors,
- content_type=content_type,
- sort_type=sort_type,
- publish_time=publish_time,
- account_id=account_id,
- limit=limit,
- settings=settings,
- http_client=client,
- rate_limiter=rate_limiter,
- env_file=env_file,
- )
- page_hits = list(page.rows)
- provider = page.provider
- current_cursor = cursors.get(provider, _douyin_cursor0(provider))
- for hit in page_hits:
- if hit.content_id not in seen:
- seen.add(hit.content_id)
- out.append(hit)
- if len(out) >= limit:
- return out
- if not page.has_more or not page.next_cursor or page.next_cursor == current_cursor:
- break
- cursors = dict(page.raw_metadata.get("cursors") or cursors)
- finally:
- if owns_client:
- client.close()
- return out[:limit]
- def search_keyword(
- keyword: str,
- *,
- platform: str = "xiaohongshu",
- content_type: str | None = "图文",
- sort_type: Optional[str] = None, # None → 平台默认(小红书:综合 / 抖音:综合排序)
- publish_time: Optional[str] = None, # None → 平台默认(抖音:不限)
- account_id: Optional[str] = None, # None → 平台默认(抖音:771431222;小红书无)
- limit: int = 5,
- settings: Optional[Settings] = None,
- http_client: Any = None,
- rate_limiter: Optional[RateLimiter] = None,
- env_file: str = ".env",
- max_pages: int = 10,
- ) -> list[str]:
- """搜索 → content_id 列表(翻页直到够 limit 或 has_more=False;按 id 去重、截断 limit)。
- 平台专属参数(path/sort_type/publish_time/起始cursor/account_id)取自 PLATFORM_SEARCH,
- 显式入参可覆盖。复用 crawler.RateLimiter('{platform}_keyword' bucket)。失败抛 SearchError。"""
- cfg = PLATFORM_SEARCH.get(platform)
- if not cfg:
- raise SearchError(f"unsupported platform: {platform}")
- settings = settings or Settings.from_env(env_file)
- is_douyin = platform == "douyin"
- if account_id is None:
- account_id = settings.piaoquantv_douyin_account_id if is_douyin else cfg["account_id"]
- if is_douyin:
- return [
- hit.content_id
- for hit in search_douyin_hits(
- keyword,
- content_type=content_type,
- sort_type=sort_type,
- publish_time=publish_time,
- account_id=account_id,
- limit=limit,
- settings=settings,
- http_client=http_client,
- rate_limiter=rate_limiter,
- env_file=env_file,
- max_pages=max_pages,
- )
- ]
- sort_type = cfg["sort_type"] if sort_type is None else sort_type
- publish_time = cfg["publish_time"] if publish_time is None else publish_time
- base_url = settings.piaoquantv_douyin_base_url if is_douyin else settings.aiddit_crawler_base_url
- rate_limiter = rate_limiter or RateLimiter(
- min_interval_seconds=CRAWLER_MIN_INTERVAL_SECONDS,
- max_interval_seconds=CRAWLER_MAX_INTERVAL_SECONDS,
- )
- owns_client = http_client is None
- client = http_client or httpx.Client()
- out: list[str] = []
- seen: set[str] = set()
- cursor = cfg["cursor0"]
- try:
- for _ in range(max_pages):
- rate_limiter.wait(f"{platform}_keyword")
- url = urljoin(base_url, cfg["path"])
- body = {"keyword": keyword,
- "sort_type": sort_type, "publish_time": publish_time, "cursor": cursor}
- if content_type:
- body["content_type"] = content_type
- if account_id:
- body["account_id"] = account_id # 抖音必带;小红书不带
- if is_douyin:
- body["cookie_batch"] = settings.piaoquantv_douyin_cookie_batch # piaoquantv 抖音必带
- try:
- resp = client.post(url, json=body,
- headers={"Content-Type": "application/json"},
- timeout=settings.crawler_timeout)
- resp.raise_for_status()
- data = resp.json()
- except httpx.HTTPError as exc:
- raise SearchError(f"http_error: {exc}") from exc
- except ValueError as exc:
- raise SearchError("bad_json") from exc
- ids, has_more, next_cursor = parse_search_response(data, platform=platform)
- for cid in ids:
- if cid not in seen:
- seen.add(cid)
- out.append(cid)
- if len(out) >= limit:
- return out
- if not has_more or not next_cursor or next_cursor == cursor:
- break
- cursor = next_cursor
- finally:
- if owns_client:
- client.close()
- return out[:limit]
- # ---------- 微信公众号搜索(回的是带 url/封面的文章,不是纯 id,故单列)----------
- _EM = re.compile(r"<[^>]+>")
- def parse_weixin(response: Any, *, limit: int = 5) -> list[dict]:
- """微信搜索回包 {code,data:{data:[{title,url,cover_url,nick_name,time}]}} → 文章列表(title 去 <em> 标记)。"""
- if not isinstance(response, dict):
- raise SearchError("bad_response: not a dict")
- if response.get("code") not in (0, "0"):
- raise SearchError(f"business_error: code={response.get('code')} msg={response.get('msg')}")
- items = ((response.get("data") or {}).get("data")) or []
- out: list[dict] = []
- for it in items:
- if not isinstance(it, dict) or not it.get("url"):
- continue
- out.append({"title": _EM.sub("", it.get("title", "")).strip(),
- "url": it.get("url", ""), "cover_url": it.get("cover_url") or "",
- "nick_name": it.get("nick_name") or "", "time": it.get("time") or ""})
- if len(out) >= limit:
- break
- return out
- def search_weixin(
- keyword: str,
- *,
- limit: int = 5,
- settings: Optional[Settings] = None,
- http_client: Any = None,
- rate_limiter: Optional[RateLimiter] = None,
- env_file: str = ".env",
- ) -> list[dict]:
- """微信公众号搜索:query → 文章列表(带 url/封面,无需 detail)。失败抛 SearchError。"""
- settings = settings or Settings.from_env(env_file)
- rate_limiter = rate_limiter or RateLimiter(
- min_interval_seconds=CRAWLER_MIN_INTERVAL_SECONDS,
- max_interval_seconds=CRAWLER_MAX_INTERVAL_SECONDS,
- )
- owns_client = http_client is None
- client = http_client or httpx.Client()
- try:
- rate_limiter.wait("weixin_keyword")
- url = urljoin(settings.aiddit_crawler_base_url, "/crawler/wei_xin/keyword")
- try:
- resp = client.post(url, json={"keyword": keyword, "cursor": "0"},
- headers={"Content-Type": "application/json"},
- timeout=settings.crawler_timeout)
- resp.raise_for_status()
- data = resp.json()
- except httpx.HTTPError as exc:
- raise SearchError(f"http_error: {exc}") from exc
- except ValueError as exc:
- raise SearchError("bad_json") from exc
- return parse_weixin(data, limit=limit)
- finally:
- if owns_client:
- client.close()
- def search_weixin_page(
- keyword: str,
- *,
- cursor: str = "0",
- limit: int = 10,
- settings: Optional[Settings] = None,
- http_client: Any = None,
- rate_limiter: Optional[RateLimiter] = None,
- env_file: str = ".env",
- ) -> SearchPage:
- """微信公众号搜索单页:query + cursor → SearchPage。"""
- settings = settings or Settings.from_env(env_file)
- rate_limiter = rate_limiter or RateLimiter(
- min_interval_seconds=CRAWLER_MIN_INTERVAL_SECONDS,
- max_interval_seconds=CRAWLER_MAX_INTERVAL_SECONDS,
- )
- owns_client = http_client is None
- client = http_client or httpx.Client()
- body = {"keyword": keyword, "cursor": cursor}
- try:
- rate_limiter.wait("weixin_keyword")
- url = urljoin(settings.aiddit_crawler_base_url, "/crawler/wei_xin/keyword")
- try:
- resp = client.post(url, json=body,
- headers={"Content-Type": "application/json"},
- timeout=settings.crawler_timeout)
- resp.raise_for_status()
- data = resp.json()
- except httpx.HTTPError as exc:
- raise SearchError(f"http_error: {exc}") from exc
- except ValueError as exc:
- raise SearchError("bad_json") from exc
- if not isinstance(data, dict):
- raise SearchError("bad_response: not a dict")
- if data.get("code") not in (0, "0"):
- raise SearchError(f"business_error: code={data.get('code')} msg={data.get('msg')}")
- envelope = data.get("data") or {}
- raw_rows = envelope.get("data") or []
- if not isinstance(raw_rows, list):
- raise SearchError("bad_response: data.data is not a list")
- return SearchPage(
- rows=parse_weixin(data, limit=limit),
- raw_count=len(raw_rows),
- cursor=cursor,
- next_cursor=str(envelope.get("next_cursor") or ""),
- has_more=bool(envelope.get("has_more")),
- provider="aiddit",
- request_payload=body,
- )
- finally:
- if owns_client:
- client.close()
- def fetch_weixin_detail(
- url: str,
- *,
- settings: Optional[Settings] = None,
- http_client: Any = None,
- rate_limiter: Optional[RateLimiter] = None,
- env_file: str = ".env",
- ) -> tuple[str, list[str]]:
- """微信公众号文章详情:content_link → (body_text, image_urls)。实测不需 token。失败抛 SearchError。"""
- settings = settings or Settings.from_env(env_file)
- rate_limiter = rate_limiter or RateLimiter(
- min_interval_seconds=CRAWLER_MIN_INTERVAL_SECONDS,
- max_interval_seconds=CRAWLER_MAX_INTERVAL_SECONDS,
- )
- owns_client = http_client is None
- client = http_client or httpx.Client()
- try:
- rate_limiter.wait("weixin_detail")
- api = urljoin(settings.aiddit_crawler_base_url, "/crawler/wei_xin/detail")
- try:
- resp = client.post(api, json={"content_link": url},
- headers={"Content-Type": "application/json"},
- timeout=settings.crawler_timeout)
- resp.raise_for_status()
- data = resp.json()
- except httpx.HTTPError as exc:
- raise SearchError(f"http_error: {exc}") from exc
- except ValueError as exc:
- raise SearchError("bad_json") from exc
- if data.get("code") not in (0, "0"):
- raise SearchError(f"business_error: code={data.get('code')} msg={data.get('msg')}")
- inner = ((data.get("data") or {}).get("data")) or {}
- imgs = []
- for im in inner.get("image_url_list") or []:
- u = im.get("image_url") if isinstance(im, dict) else im
- if u:
- imgs.append(u)
- return inner.get("body_text") or "", imgs
- finally:
- if owns_client:
- client.close()
- # ---------- 小红书搜索(封面/标题/作者已在搜索回包的 note_card 里,无需 detail)----------
- def parse_xiaohongshu(response: Any, *, limit: int = 5) -> list[dict]:
- """小红书搜索回包 {code,data:{data:[{id, note_card:{image_list,display_title,desc,user,type}}]}}
- → 帖子列表,直接取封面(image_list[0].image_url)+标题+作者,省去逐条 detail。无视频直链。"""
- if not isinstance(response, dict):
- raise SearchError("bad_response: not a dict")
- if response.get("code") not in (0, "0"):
- raise SearchError(f"business_error: code={response.get('code')} msg={response.get('msg')}")
- items = ((response.get("data") or {}).get("data")) or []
- out: list[dict] = []
- for it in items:
- if not isinstance(it, dict):
- continue
- cid = it.get("id")
- nc = it.get("note_card") or {}
- img_list = nc.get("image_list") or []
- cover = (img_list[0] or {}).get("image_url") if img_list else ""
- if not cid or not cover:
- continue
- title = (nc.get("display_title") or "").strip()
- if not title:
- title = clip_text((nc.get("desc") or "").strip().split("\n")[0], TITLE_FALLBACK_MAX_CHARS)
- out.append({"id": cid, "title": title, "cover_url": cover,
- "nick_name": (nc.get("user") or {}).get("nickname") or "",
- "type": nc.get("type") or "",
- "url": f"https://www.xiaohongshu.com/explore/{cid}"})
- if len(out) >= limit:
- break
- return out
- def search_xiaohongshu(
- keyword: str,
- *,
- content_type: str | None = None,
- limit: int = 5,
- settings: Optional[Settings] = None,
- http_client: Any = None,
- rate_limiter: Optional[RateLimiter] = None,
- env_file: str = ".env",
- ) -> list[dict]:
- """小红书搜索:query → 帖子列表(带 url/封面/标题,无需 detail;视频帖无直链)。失败抛 SearchError。"""
- settings = settings or Settings.from_env(env_file)
- rate_limiter = rate_limiter or RateLimiter(
- min_interval_seconds=CRAWLER_MIN_INTERVAL_SECONDS,
- max_interval_seconds=CRAWLER_MAX_INTERVAL_SECONDS,
- )
- owns_client = http_client is None
- client = http_client or httpx.Client()
- try:
- rate_limiter.wait("xiaohongshu_keyword")
- url = urljoin(settings.aiddit_crawler_base_url, "/crawler/xiao_hong_shu/keyword")
- try:
- body = {"keyword": keyword, "sort_type": "综合", "publish_time": "", "cursor": ""}
- if content_type:
- body["content_type"] = content_type
- resp = client.post(url, json=body,
- headers={"Content-Type": "application/json"},
- timeout=settings.crawler_timeout)
- resp.raise_for_status()
- data = resp.json()
- except httpx.HTTPError as exc:
- raise SearchError(f"http_error: {exc}") from exc
- except ValueError as exc:
- raise SearchError("bad_json") from exc
- return parse_xiaohongshu(data, limit=limit)
- finally:
- if owns_client:
- client.close()
- def search_xiaohongshu_page(
- keyword: str,
- *,
- cursor: str = "",
- content_type: str | None = None,
- limit: int = 10,
- settings: Optional[Settings] = None,
- http_client: Any = None,
- rate_limiter: Optional[RateLimiter] = None,
- env_file: str = ".env",
- ) -> SearchPage:
- """小红书搜索单页:query + cursor → SearchPage。"""
- settings = settings or Settings.from_env(env_file)
- rate_limiter = rate_limiter or RateLimiter(
- min_interval_seconds=CRAWLER_MIN_INTERVAL_SECONDS,
- max_interval_seconds=CRAWLER_MAX_INTERVAL_SECONDS,
- )
- owns_client = http_client is None
- client = http_client or httpx.Client()
- body = {"keyword": keyword, "sort_type": "综合", "publish_time": "", "cursor": cursor}
- if content_type:
- body["content_type"] = content_type
- try:
- rate_limiter.wait("xiaohongshu_keyword")
- url = urljoin(settings.aiddit_crawler_base_url, "/crawler/xiao_hong_shu/keyword")
- try:
- resp = client.post(url, json=body,
- headers={"Content-Type": "application/json"},
- timeout=settings.crawler_timeout)
- resp.raise_for_status()
- data = resp.json()
- except httpx.HTTPError as exc:
- raise SearchError(f"http_error: {exc}") from exc
- except ValueError as exc:
- raise SearchError("bad_json") from exc
- if not isinstance(data, dict):
- raise SearchError("bad_response: not a dict")
- if data.get("code") not in (0, "0"):
- raise SearchError(f"business_error: code={data.get('code')} msg={data.get('msg')}")
- envelope = data.get("data") or {}
- raw_rows = envelope.get("data") or []
- if not isinstance(raw_rows, list):
- raise SearchError("bad_response: data.data is not a list")
- return SearchPage(
- rows=parse_xiaohongshu(data, limit=limit),
- raw_count=len(raw_rows),
- cursor=cursor,
- next_cursor=str(envelope.get("next_cursor") or ""),
- has_more=bool(envelope.get("has_more")),
- provider="aiddit",
- request_payload=body,
- )
- finally:
- if owns_client:
- client.close()
|