Files
holy-python/crawler_client.py
2026-08-04 14:02:45 +08:00

116 lines
4.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import re
import time
from typing import Any
import requests
from config import Config
class CrawlerError(RuntimeError):
pass
CRAWLER_SESSION = requests.Session()
# 爬虫服务为明确配置的直连地址,不应继承 macOS 系统代理。
# 否则 requests 会经由 127.0.0.1:7890约 30 秒后被代理返回 502。
CRAWLER_SESSION.trust_env = False
def normalize_collection_keywords(keywords: list[str]) -> list[str]:
"""清理影响小红书问一问卡片匹配的末尾标点,并保持关键词顺序。"""
normalized: list[str] = []
for keyword in keywords:
value = re.sub(r"[?!!。..]+$", "", str(keyword).strip()).strip()
if value and value not in normalized:
normalized.append(value)
if not normalized:
raise CrawlerError("关键词不能为空")
return normalized
def _post(
path: str,
body: dict[str, Any],
*,
retry_transient: bool = True,
) -> dict[str, Any]:
url = f"{Config.CRAWLER_BASE_URL}{path}"
retry_times = max(1, Config.CRAWLER_RETRY_TIMES) if retry_transient else 1
response: requests.Response | None = None
for attempt in range(1, retry_times + 1):
try:
response = CRAWLER_SESSION.post(
url,
json=body,
timeout=Config.CRAWLER_TIMEOUT_SECONDS,
)
# 网关临时异常时自动重试,避免直接将业务任务标记为失败。
if response.status_code in {502, 503, 504} and attempt < retry_times:
time.sleep(Config.CRAWLER_RETRY_DELAY_SECONDS * attempt)
continue
response.raise_for_status()
payload = response.json()
break
except (requests.ConnectionError, requests.Timeout) as exc:
if attempt < retry_times:
time.sleep(Config.CRAWLER_RETRY_DELAY_SECONDS * attempt)
continue
raise CrawlerError(f"爬虫服务请求失败(已重试 {retry_times} 次):{exc}") from exc
except requests.RequestException as exc:
detail = response.text[:500] if response is not None else ""
message = f"爬虫服务请求失败:{exc}"
if detail:
message += f",响应:{detail}"
raise CrawlerError(message) from exc
except ValueError as exc:
raise CrawlerError("爬虫服务返回的不是有效 JSON") from exc
else:
raise CrawlerError(f"爬虫服务请求失败(已重试 {retry_times} 次)")
if payload.get("code") != 200 or not payload.get("success", False):
raise CrawlerError(payload.get("msg") or "爬虫服务返回失败")
data = payload.get("data")
if not isinstance(data, dict):
raise CrawlerError("爬虫服务响应缺少 data")
return data
def start_collection(mode: str, keywords: list[str]) -> dict[str, Any]:
"""深度任务走 sources轻度/常规任务走 overview。"""
collection_keywords = normalize_collection_keywords(keywords)
path = "/api/v1/xhs/wen/sources" if mode == "deep" else "/api/v1/xhs/wen/overview"
# 远程 overview 只有接收数组时才创建异步任务并立即返回 task_id。
# 因此轻度任务即使只有一个关键词也必须传数组,避免单关键词走同步
# overview 并因采集耗时超过网关时限而返回 502。
keyword: str | list[str] = (
collection_keywords if mode == "mild" else collection_keywords[0]
)
# 轻度 overview 在首次计算超过网关时限后,远程通常已经写好缓存,
# 允许受控重试以取得结果;深度 sources 会启动长任务,禁止盲目重试,
# 避免远程请求虽断开但后台继续执行时产生重复采集任务。
return _post(
path,
{"keyword": keyword},
retry_transient=(mode != "deep"),
)
def query_collection(crawler_task_id: str) -> dict[str, Any]:
return _post("/api/v1/xhs/wen/tasks/query", {"task_id": crawler_task_id})
def cancel_collection(crawler_task_id: str) -> dict[str, Any]:
"""通知爬虫任务在下一个安全检查点停止并保存部分结果。"""
return _post(f"/api/v1/xhs/wen/tasks/{crawler_task_id}/cancel", {})
def resume_collection(crawler_task_id: str) -> dict[str, Any]:
"""沿用原爬虫任务 ID从 framework 保存的检查点继续采集。"""
return _post(f"/api/v1/xhs/wen/tasks/{crawler_task_id}/resume", {})