Files
Agent-Reach/agent_reach/channels/xiaohongshu.py
T
Pnant bc2d554e13 fix(xiaohongshu): switch from Docker MCP to xhs-cli (pipx install) (#236)
Docker + Cookie import → `pipx install xiaohongshu-cli` + `xhs login`.
Tier downgraded from 2 to 1. xhs-cli (1,477 stars) auto-extracts
cookies from browser, supports search/read/comment/post/hot/feed.

- xiaohongshu.py: check() detects `xhs` binary, parses `xhs status`
- cli.py: install via pipx instead of Docker
- skill/references/social.md: updated XHS commands
- tests: 3 new tests for CLI-based check()
- 77 tests passing

Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-03-31 17:17:19 +08:00

163 lines
5.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""XiaoHongShu — check if xhs-cli (xiaohongshu-cli) is available."""
import shutil
import subprocess
from .base import Channel
def format_xhs_result(data):
"""Clean XHS API response, keeping only useful fields.
Handles both single note objects and lists of notes (search results).
Drastically reduces token usage by stripping structural redundancy (#134).
"""
if isinstance(data, list):
return [_clean_note(item) for item in data]
if isinstance(data, dict):
# Handle search_feeds wrapper: {"items": [...]} or {"data": {"items": [...]}}
items = None
if "items" in data:
items = data["items"]
elif "data" in data and isinstance(data.get("data"), dict):
items = data["data"].get("items") or data["data"].get("notes")
if items and isinstance(items, list):
return [_clean_note(item) for item in items]
# Single note
return _clean_note(data)
return data
def _clean_note(note):
"""Extract useful fields from a single XHS note/feed item."""
if not isinstance(note, dict):
return note
# Some responses nest the note under "note_card" or "note"
inner = note.get("note_card") or note.get("note") or note
result = {}
# Basic info
for key in ("id", "note_id", "xsec_token", "title", "desc", "type", "time"):
if key in inner:
result[key] = inner[key]
# Content (may be in desc or content)
if "content" in inner and "desc" not in result:
result["content"] = inner["content"]
# Author
user = inner.get("user") or inner.get("author")
if isinstance(user, dict):
result["user"] = {
k: user[k] for k in ("nickname", "user_id", "nick_name") if k in user
}
# Engagement metrics
interact = inner.get("interact_info") or inner.get("note_interact_info") or {}
if isinstance(interact, dict):
for key in ("liked_count", "collected_count", "comment_count", "share_count"):
if key in interact:
result[key] = interact[key]
# Also check top-level (some API formats)
for key in ("liked_count", "collected_count", "comment_count", "share_count"):
if key in inner and key not in result:
result[key] = inner[key]
# Images — just URLs
images = inner.get("image_list") or inner.get("images_list") or []
if isinstance(images, list):
urls = []
for img in images:
if isinstance(img, dict):
url = img.get("url") or img.get("url_default") or img.get("original")
if url:
urls.append(url)
elif isinstance(img, str):
urls.append(img)
if urls:
result["images"] = urls
# Tags
tags = inner.get("tag_list") or inner.get("tags") or []
if isinstance(tags, list):
tag_names = []
for t in tags:
if isinstance(t, dict) and "name" in t:
tag_names.append(t["name"])
elif isinstance(t, str):
tag_names.append(t)
if tag_names:
result["tags"] = tag_names
# Comments (if present, e.g. from get_feed_detail with comments)
comments = inner.get("comments") or []
if isinstance(comments, list) and comments:
result["comments"] = [_clean_comment(c) for c in comments]
return result
def _clean_comment(comment):
"""Extract useful fields from a comment."""
if not isinstance(comment, dict):
return comment
result = {}
if "content" in comment:
result["content"] = comment["content"]
user = comment.get("user_info") or comment.get("user")
if isinstance(user, dict):
result["user"] = user.get("nickname") or user.get("nick_name", "")
for key in ("like_count", "sub_comment_count"):
if key in comment:
result[key] = comment[key]
return result
class XiaoHongShuChannel(Channel):
name = "xiaohongshu"
description = "小红书笔记"
backends = ["xhs-cli (xiaohongshu-cli)"]
tier = 1
def can_handle(self, url: str) -> bool:
from urllib.parse import urlparse
d = urlparse(url).netloc.lower()
return "xiaohongshu.com" in d or "xhslink.com" in d
def check(self, config=None):
xhs = shutil.which("xhs")
if not xhs:
return "off", (
"需要安装 xhs-cli\n"
" pipx install xiaohongshu-cli\n"
"或:\n"
" uv tool install xiaohongshu-cli\n"
"安装后运行 `xhs login` 登录"
)
try:
r = subprocess.run(
[xhs, "status"], capture_output=True,
encoding="utf-8", errors="replace", timeout=10,
)
output = (r.stdout or "") + (r.stderr or "")
if r.returncode == 0 and "ok: true" in output:
return "ok", (
"完整可用(搜索、阅读、评论、发帖、热门、"
"收藏、关注、用户查询)"
)
if "not_authenticated" in output or "expired" in output:
return "warn", (
"xhs-cli 已安装但未登录。运行:\n"
" xhs login\n"
"(自动从浏览器提取 Cookie,或扫码登录)"
)
return "warn", (
"xhs-cli 已安装但状态异常。运行:\n"
" xhs -v status 查看详细信息"
)
except Exception:
return "warn", "xhs-cli 已安装但连接失败"