Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
67 changes: 51 additions & 16 deletions agent_reach/channels/xiaohongshu.py
Original file line number Diff line number Diff line change
Expand Up @@ -57,47 +57,76 @@ def format_xhs_result(data):
if isinstance(data, list):
return [_clean_note(item) for item in data]
if isinstance(data, dict):
# Handle search_feeds wrapper: {"items": [...]} or {"data": {"items": [...]}}
items = None
if "items" in data:
items = data["items"]
elif "data" in data and isinstance(data.get("data"), dict):
items = data["data"].get("items") or data["data"].get("notes")
if items and isinstance(items, list):
return [_clean_note(item) for item in items]
# Single note
return _clean_note(data)
# The MCP CLI emits {"feeds": [...]}; HTTP adds a "data" envelope.
payload = data.get("data")
if not isinstance(payload, dict):
payload = data
for key in ("items", "notes", "feeds"):
if key in payload and isinstance(payload[key], list):
return [_clean_note(item) for item in payload[key]]
return _clean_note(payload)
return data


# Public xiaohongshu-mcp uses camelCase; keep the formatter's existing
# snake_case output contract and support older snake_case backends as well.
_XHS_FIELD_ALIASES = {
"noteCard": "note_card", "noteId": "note_id", "xsecToken": "xsec_token",
"displayTitle": "title", "userId": "user_id", "nickName": "nick_name",
"interactInfo": "interact_info", "likedCount": "liked_count",
"collectedCount": "collected_count", "commentCount": "comment_count",
"sharedCount": "share_count", "imageList": "image_list",
"urlDefault": "url_default", "userInfo": "user_info",
"likeCount": "like_count", "subCommentCount": "sub_comment_count",
}


def _normalize_xhs_fields(value):
"""Add known aliases without mutating input or replacing canonical fields."""
if not isinstance(value, dict):
return value
result = dict(value)
for alias, canonical in _XHS_FIELD_ALIASES.items():
if alias in value and canonical not in result:
result[canonical] = value[alias]
return result


def _clean_note(note):
"""Extract useful fields from a single XHS note/feed item."""
if not isinstance(note, dict):
return note

# Some responses nest the note under "note_card" or "note"
inner = note.get("note_card") or note.get("note") or note
note = _normalize_xhs_fields(note)
# Search identity is on the feed, while content is inside noteCard.
inner = _normalize_xhs_fields(note.get("note_card") or note.get("note") or note)
if not isinstance(inner, dict):
return {}

result = {}

# Basic info
for key in ("id", "note_id", "xsec_token", "title", "desc", "type", "time"):
if key in inner:
result[key] = inner[key]
elif key in ("id", "note_id", "xsec_token") and key in note:
result[key] = note[key]

# Content (may be in desc or content)
if "content" in inner and "desc" not in result:
result["content"] = inner["content"]

# Author
user = inner.get("user") or inner.get("author")
user = _normalize_xhs_fields(inner.get("user") or inner.get("author"))
if isinstance(user, dict):
result["user"] = {
k: user[k] for k in ("nickname", "user_id", "nick_name") if k in user
}

# Engagement metrics
interact = inner.get("interact_info") or inner.get("note_interact_info") or {}
interact = _normalize_xhs_fields(
inner.get("interact_info") or inner.get("note_interact_info") or {}
)
if isinstance(interact, dict):
for key in ("liked_count", "collected_count", "comment_count", "share_count"):
if key in interact:
Expand All @@ -109,9 +138,12 @@ def _clean_note(note):

# Images — just URLs
images = inner.get("image_list") or inner.get("images_list") or []
if not images and isinstance(inner.get("cover"), dict):
images = [inner["cover"]]
if isinstance(images, list):
urls = []
for img in images:
img = _normalize_xhs_fields(img)
if isinstance(img, dict):
url = img.get("url") or img.get("url_default") or img.get("original")
if url:
Expand All @@ -134,7 +166,9 @@ def _clean_note(note):
result["tags"] = tag_names

# Comments (if present, e.g. from get_feed_detail with comments)
comments = inner.get("comments") or []
comments = inner.get("comments") or note.get("comments") or []
if isinstance(comments, dict):
comments = comments.get("list", [])
if isinstance(comments, list) and comments:
result["comments"] = [_clean_comment(c) for c in comments]

Expand All @@ -145,10 +179,11 @@ def _clean_comment(comment):
"""Extract useful fields from a comment."""
if not isinstance(comment, dict):
return comment
comment = _normalize_xhs_fields(comment)
result = {}
if "content" in comment:
result["content"] = comment["content"]
user = comment.get("user_info") or comment.get("user")
user = _normalize_xhs_fields(comment.get("user_info") or comment.get("user"))
if isinstance(user, dict):
result["user"] = user.get("nickname") or user.get("nick_name", "")
for key in ("like_count", "sub_comment_count"):
Expand Down
129 changes: 129 additions & 0 deletions tests/test_xhs_mcp_format.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,129 @@
"""Exercise the public xiaohongshu-mcp schema through the formatter CLI.

Field names follow xpzouying/xiaohongshu-mcp's service.go FeedsListResponse
and xiaohongshu/types.go Feed, FeedDetailResponse, and CommentList structs.
"""

import json
import subprocess
import sys

import pytest

from agent_reach.channels.xiaohongshu import format_xhs_result
from agent_reach.utils.process import utf8_subprocess_env

MCP_FEED = {
"id": "note-123",
"xsecToken": "explicit-test-token",
"modelType": "note",
"noteCard": {
"type": "normal",
"displayTitle": "搜索结果",
"user": {"userId": "user-1", "nickname": "作者", "avatar": "discard"},
"interactInfo": {"likedCount": "12", "commentCount": "3", "sharedCount": "2"},
"cover": {"urlDefault": "https://example.com/cover.jpg", "width": 1080},
},
}


def _format_cli(payload):
result = subprocess.run(
[sys.executable, "-m", "agent_reach.cli", "format", "xhs"],
input=json.dumps(payload, ensure_ascii=False),
capture_output=True,
encoding="utf-8",
env=utf8_subprocess_env(),
timeout=10,
check=True,
)
return json.loads(result.stdout)


def test_mcp_search_cli_preserves_fields_needed_for_detail():
result = _format_cli({"feeds": [MCP_FEED], "count": 1})
assert result == [
{
"id": "note-123",
"xsec_token": "explicit-test-token",
"title": "搜索结果",
"type": "normal",
"user": {"nickname": "作者", "user_id": "user-1"},
"liked_count": "12",
"comment_count": "3",
"share_count": "2",
"images": ["https://example.com/cover.jpg"],
}
]


def test_mcp_http_success_wrapper_formats_the_same_search():
wrapped = {"success": True, "data": {"feeds": [MCP_FEED], "count": 1}}
assert format_xhs_result(wrapped) == format_xhs_result([MCP_FEED])
assert format_xhs_result(wrapped)[0]["title"] == "搜索结果"


def test_mcp_detail_cli_retains_note_and_sibling_comments():
payload = {
"feed_id": "note-123",
"data": {
"note": {
"noteId": "note-123",
"xsecToken": "explicit-test-token",
"title": "详情",
"desc": "正文",
"type": "normal",
"user": {"userId": "user-1", "nickname": "作者"},
"interactInfo": {"collectedCount": "4"},
"imageList": [{"urlDefault": "https://example.com/detail.jpg"}],
},
"comments": {
"list": [
{
"content": "好",
"userInfo": {"nickname": "读者"},
"likeCount": "2",
"subCommentCount": "0",
}
],
"cursor": "discard",
"hasMore": False,
},
},
}
result = _format_cli(payload)
assert result["note_id"] == "note-123"
assert result["xsec_token"] == "explicit-test-token"
assert result["title"] == "详情"
assert result["desc"] == "正文"
assert result["collected_count"] == "4"
assert result["images"] == ["https://example.com/detail.jpg"]
assert result["comments"] == [
{"content": "好", "user": "读者", "like_count": "2", "sub_comment_count": "0"}
]


@pytest.mark.parametrize(
"payload",
[
{"feeds": [], "count": 0},
{"items": []},
{"data": {"items": []}},
{"data": {"notes": []}},
{"data": {"items": [], "notes": [MCP_FEED]}},
],
)
def test_empty_search_remains_a_list(payload):
assert format_xhs_result(payload) == []


def test_snake_case_card_retains_outer_identity():
result = format_xhs_result(
{"id": "outer", "xsec_token": "token", "note_card": {"title": "card"}}
)
assert result == {"id": "outer", "xsec_token": "token", "title": "card"}


def test_inner_identity_keeps_precedence_for_legacy_wrappers():
result = format_xhs_result({"id": "outer", "note_card": {"id": "inner"}})
assert result["id"] == "inner"