From fe8b16343e52ce10abc196efd29408fabec4d088 Mon Sep 17 00:00:00 2001 From: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:15:06 -0400 Subject: [PATCH 1/2] fix(xhs): format public MCP search and detail responses Signed-off-by: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> --- agent_reach/channels/xiaohongshu.py | 67 +++++++++++---- tests/test_xhs_mcp_format.py | 127 ++++++++++++++++++++++++++++ 2 files changed, 178 insertions(+), 16 deletions(-) create mode 100644 tests/test_xhs_mcp_format.py diff --git a/agent_reach/channels/xiaohongshu.py b/agent_reach/channels/xiaohongshu.py index a72dd0bf5..86a0e2fb7 100644 --- a/agent_reach/channels/xiaohongshu.py +++ b/agent_reach/channels/xiaohongshu.py @@ -57,26 +57,51 @@ def format_xhs_result(data): if isinstance(data, list): return [_clean_note(item) for item in data] if isinstance(data, dict): - # Handle search_feeds wrapper: {"items": [...]} or {"data": {"items": [...]}} - items = None - if "items" in data: - items = data["items"] - elif "data" in data and isinstance(data.get("data"), dict): - items = data["data"].get("items") or data["data"].get("notes") - if items and isinstance(items, list): - return [_clean_note(item) for item in items] - # Single note - return _clean_note(data) + # The MCP CLI emits {"feeds": [...]}; HTTP adds a "data" envelope. + payload = data.get("data") + if not isinstance(payload, dict): + payload = data + for key in ("items", "notes", "feeds"): + if key in payload and isinstance(payload[key], list): + return [_clean_note(item) for item in payload[key]] + return _clean_note(payload) return data +# Public xiaohongshu-mcp uses camelCase; keep the formatter's existing +# snake_case output contract and support older snake_case backends as well. +_XHS_FIELD_ALIASES = { + "noteCard": "note_card", "noteId": "note_id", "xsecToken": "xsec_token", + "displayTitle": "title", "userId": "user_id", "nickName": "nick_name", + "interactInfo": "interact_info", "likedCount": "liked_count", + "collectedCount": "collected_count", "commentCount": "comment_count", + "sharedCount": "share_count", "imageList": "image_list", + "urlDefault": "url_default", "userInfo": "user_info", + "likeCount": "like_count", "subCommentCount": "sub_comment_count", +} + + +def _normalize_xhs_fields(value): + """Add known aliases without mutating input or replacing canonical fields.""" + if not isinstance(value, dict): + return value + result = dict(value) + for alias, canonical in _XHS_FIELD_ALIASES.items(): + if alias in value and canonical not in result: + result[canonical] = value[alias] + return result + + def _clean_note(note): """Extract useful fields from a single XHS note/feed item.""" if not isinstance(note, dict): return note - # Some responses nest the note under "note_card" or "note" - inner = note.get("note_card") or note.get("note") or note + note = _normalize_xhs_fields(note) + # Search identity is on the feed, while content is inside noteCard. + inner = _normalize_xhs_fields(note.get("note_card") or note.get("note") or note) + if not isinstance(inner, dict): + return {} result = {} @@ -84,20 +109,24 @@ def _clean_note(note): for key in ("id", "note_id", "xsec_token", "title", "desc", "type", "time"): if key in inner: result[key] = inner[key] + elif key in ("id", "note_id", "xsec_token") and key in note: + result[key] = note[key] # Content (may be in desc or content) if "content" in inner and "desc" not in result: result["content"] = inner["content"] # Author - user = inner.get("user") or inner.get("author") + user = _normalize_xhs_fields(inner.get("user") or inner.get("author")) if isinstance(user, dict): result["user"] = { k: user[k] for k in ("nickname", "user_id", "nick_name") if k in user } # Engagement metrics - interact = inner.get("interact_info") or inner.get("note_interact_info") or {} + interact = _normalize_xhs_fields( + inner.get("interact_info") or inner.get("note_interact_info") or {} + ) if isinstance(interact, dict): for key in ("liked_count", "collected_count", "comment_count", "share_count"): if key in interact: @@ -109,9 +138,12 @@ def _clean_note(note): # Images — just URLs images = inner.get("image_list") or inner.get("images_list") or [] + if not images and isinstance(inner.get("cover"), dict): + images = [inner["cover"]] if isinstance(images, list): urls = [] for img in images: + img = _normalize_xhs_fields(img) if isinstance(img, dict): url = img.get("url") or img.get("url_default") or img.get("original") if url: @@ -134,7 +166,9 @@ def _clean_note(note): result["tags"] = tag_names # Comments (if present, e.g. from get_feed_detail with comments) - comments = inner.get("comments") or [] + comments = inner.get("comments") or note.get("comments") or [] + if isinstance(comments, dict): + comments = comments.get("list", []) if isinstance(comments, list) and comments: result["comments"] = [_clean_comment(c) for c in comments] @@ -145,10 +179,11 @@ def _clean_comment(comment): """Extract useful fields from a comment.""" if not isinstance(comment, dict): return comment + comment = _normalize_xhs_fields(comment) result = {} if "content" in comment: result["content"] = comment["content"] - user = comment.get("user_info") or comment.get("user") + user = _normalize_xhs_fields(comment.get("user_info") or comment.get("user")) if isinstance(user, dict): result["user"] = user.get("nickname") or user.get("nick_name", "") for key in ("like_count", "sub_comment_count"): diff --git a/tests/test_xhs_mcp_format.py b/tests/test_xhs_mcp_format.py new file mode 100644 index 000000000..bfef96962 --- /dev/null +++ b/tests/test_xhs_mcp_format.py @@ -0,0 +1,127 @@ +"""Exercise the public xiaohongshu-mcp schema through the formatter CLI. + +Field names follow xpzouying/xiaohongshu-mcp's service.go FeedsListResponse +and xiaohongshu/types.go Feed, FeedDetailResponse, and CommentList structs. +""" + +import json +import subprocess +import sys + +import pytest + +from agent_reach.channels.xiaohongshu import format_xhs_result + +MCP_FEED = { + "id": "note-123", + "xsecToken": "explicit-test-token", + "modelType": "note", + "noteCard": { + "type": "normal", + "displayTitle": "搜索结果", + "user": {"userId": "user-1", "nickname": "作者", "avatar": "discard"}, + "interactInfo": {"likedCount": "12", "commentCount": "3", "sharedCount": "2"}, + "cover": {"urlDefault": "https://example.com/cover.jpg", "width": 1080}, + }, +} + + +def _format_cli(payload): + result = subprocess.run( + [sys.executable, "-m", "agent_reach.cli", "format", "xhs"], + input=json.dumps(payload, ensure_ascii=False), + capture_output=True, + text=True, + timeout=10, + check=True, + ) + return json.loads(result.stdout) + + +def test_mcp_search_cli_preserves_fields_needed_for_detail(): + result = _format_cli({"feeds": [MCP_FEED], "count": 1}) + assert result == [ + { + "id": "note-123", + "xsec_token": "explicit-test-token", + "title": "搜索结果", + "type": "normal", + "user": {"nickname": "作者", "user_id": "user-1"}, + "liked_count": "12", + "comment_count": "3", + "share_count": "2", + "images": ["https://example.com/cover.jpg"], + } + ] + + +def test_mcp_http_success_wrapper_formats_the_same_search(): + wrapped = {"success": True, "data": {"feeds": [MCP_FEED], "count": 1}} + assert format_xhs_result(wrapped) == format_xhs_result([MCP_FEED]) + assert format_xhs_result(wrapped)[0]["title"] == "搜索结果" + + +def test_mcp_detail_cli_retains_note_and_sibling_comments(): + payload = { + "feed_id": "note-123", + "data": { + "note": { + "noteId": "note-123", + "xsecToken": "explicit-test-token", + "title": "详情", + "desc": "正文", + "type": "normal", + "user": {"userId": "user-1", "nickname": "作者"}, + "interactInfo": {"collectedCount": "4"}, + "imageList": [{"urlDefault": "https://example.com/detail.jpg"}], + }, + "comments": { + "list": [ + { + "content": "好", + "userInfo": {"nickname": "读者"}, + "likeCount": "2", + "subCommentCount": "0", + } + ], + "cursor": "discard", + "hasMore": False, + }, + }, + } + result = _format_cli(payload) + assert result["note_id"] == "note-123" + assert result["xsec_token"] == "explicit-test-token" + assert result["title"] == "详情" + assert result["desc"] == "正文" + assert result["collected_count"] == "4" + assert result["images"] == ["https://example.com/detail.jpg"] + assert result["comments"] == [ + {"content": "好", "user": "读者", "like_count": "2", "sub_comment_count": "0"} + ] + + +@pytest.mark.parametrize( + "payload", + [ + {"feeds": [], "count": 0}, + {"items": []}, + {"data": {"items": []}}, + {"data": {"notes": []}}, + {"data": {"items": [], "notes": [MCP_FEED]}}, + ], +) +def test_empty_search_remains_a_list(payload): + assert format_xhs_result(payload) == [] + + +def test_snake_case_card_retains_outer_identity(): + result = format_xhs_result( + {"id": "outer", "xsec_token": "token", "note_card": {"title": "card"}} + ) + assert result == {"id": "outer", "xsec_token": "token", "title": "card"} + + +def test_inner_identity_keeps_precedence_for_legacy_wrappers(): + result = format_xhs_result({"id": "outer", "note_card": {"id": "inner"}}) + assert result["id"] == "inner" From 9ca7034af833f854354671a3f4409a90e4a6a85b Mon Sep 17 00:00:00 2001 From: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:42:45 -0400 Subject: [PATCH 2/2] test(xhs): make native formatter fixtures use UTF-8 Signed-off-by: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> --- tests/test_xhs_mcp_format.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/test_xhs_mcp_format.py b/tests/test_xhs_mcp_format.py index bfef96962..039a9f6a8 100644 --- a/tests/test_xhs_mcp_format.py +++ b/tests/test_xhs_mcp_format.py @@ -11,6 +11,7 @@ import pytest from agent_reach.channels.xiaohongshu import format_xhs_result +from agent_reach.utils.process import utf8_subprocess_env MCP_FEED = { "id": "note-123", @@ -31,7 +32,8 @@ def _format_cli(payload): [sys.executable, "-m", "agent_reach.cli", "format", "xhs"], input=json.dumps(payload, ensure_ascii=False), capture_output=True, - text=True, + encoding="utf-8", + env=utf8_subprocess_env(), timeout=10, check=True, )