|
|
@@ -47,8 +47,7 @@ XHS_SKILLS_PATH = r"C:\code\XiaohongshuSkills\scripts"
|
|
|
sys.path.insert(0, XHS_SKILLS_PATH)
|
|
|
|
|
|
CREATOR_HOME = "https://creator.xiaohongshu.com/new/home"
|
|
|
-NOTES_MANAGE_URL = "https://creator.xiaohongshu.com/publish/publish_manage"
|
|
|
-COMMENTS_URL = "https://creator.xiaohongshu.com/message/comment"
|
|
|
+NOTES_MANAGE_URL = "https://creator.xiaohongshu.com/new/note-manager"
|
|
|
|
|
|
|
|
|
def _log(msg):
|
|
|
@@ -97,137 +96,125 @@ def _clean_title(title):
|
|
|
|
|
|
|
|
|
def collect_overview(publisher) -> dict:
|
|
|
- """从创作者中心首页获取账号总览数据。"""
|
|
|
+ """
|
|
|
+ 从创作者中心首页 /new/home 获取账号总览数据。
|
|
|
+
|
|
|
+ 页面 innerText 结构(实测):
|
|
|
+ - 账号级字段: "数字\\n标签"(如 "20\\n粉丝数", "551\\n获赞与收藏")
|
|
|
+ - 周期数据: "标签\\n数字\\n环比xx%"(如 "观看数\\n209\\n环比-86%")
|
|
|
+ - 默认显示近7日,点击「近30日」tab 抓第二组
|
|
|
+
|
|
|
+ 返回 dict:
|
|
|
+ fans_count / follow_count / interact_count(账号级)
|
|
|
+ period_7d / period_30d: 各含 exposure/view/like/comment/collect/share/net_fans
|
|
|
+ """
|
|
|
overview = {}
|
|
|
- text = (publisher._evaluate("document.body.innerText") or "")
|
|
|
|
|
|
- # 粉丝数 / 总阅读 / 总互动 / 笔记总数(宽泛匹配,兼容页面 word 差异)
|
|
|
- patterns = [
|
|
|
- ("fans_count", r"粉丝数\s*[::\n]\s*([\d.,万]+)"),
|
|
|
- ("total_read", r"总阅读[量]?\s*[::\n]\s*([\d.,万]+)"),
|
|
|
- ("total_interact", r"总互动[量]?\s*[::\n]\s*([\d.,万]+)"),
|
|
|
- ("note_count", r"笔记总数\s*[::\n]\s*([\d.,]+)"),
|
|
|
- ("fans_growth_30d", r"(?:近30天|近3[0-9]天)[^\d\n]*?([+-]?[\d.,万]+)"),
|
|
|
- ]
|
|
|
- for key, pat in patterns:
|
|
|
- m = re.search(pat, text)
|
|
|
+ def _parse_period(text):
|
|
|
+ """解析周期数据(标签\\n数字\\n环比%)。返回 dict[label] = (值, 原始字符串)"""
|
|
|
+ p = {}
|
|
|
+ # 匹配 "标签\n数字\n环比xx%" 或 "标签\n数字\n环比-"
|
|
|
+ pattern = re.compile(
|
|
|
+ r"^(曝光数|观看数|封面点击率|视频完播率|点赞数|评论数|收藏数|分享数|净涨粉|新增关注|取消关注|主页访客)"
|
|
|
+ r"\n([\d.,万%]+)\n环比([+-]?\d*%?)", re.M)
|
|
|
+ for m in pattern.finditer(text):
|
|
|
+ p[m.group(1)] = m.group(2)
|
|
|
+ return p
|
|
|
+
|
|
|
+ # --- 账号级字段:数字在标签前 ---
|
|
|
+ text = (publisher._evaluate("document.body.innerText") or "")
|
|
|
+ for key, label in (("fans_count", "粉丝数"), ("follow_count", "关注数"),
|
|
|
+ ("interact_count", "获赞与收藏")):
|
|
|
+ m = re.search(r"([\d.,万]+)\n" + re.escape(label), text)
|
|
|
if m:
|
|
|
overview[key] = _to_int(m.group(1))
|
|
|
- return overview
|
|
|
+ else:
|
|
|
+ overview[key] = 0
|
|
|
|
|
|
+ # --- 近7日(默认 tab) ---
|
|
|
+ period_7d = _parse_period(text)
|
|
|
|
|
|
-def collect_notes_from_dom(publisher, limit=10):
|
|
|
- """从笔记管理页 DOM 提取笔记列表(完整标题 + 数据)。"""
|
|
|
- expr = f"""() => {{
|
|
|
- const notes = [];
|
|
|
- // 常见容器选择器:数据表格行 / 笔记卡片
|
|
|
- const rows = document.querySelectorAll(
|
|
|
- 'table tbody tr, .note-item, [class*="note-row"], [class*="note-item"], [class*="data-card"], li'
|
|
|
- );
|
|
|
- let count = 0;
|
|
|
- for (const row of rows) {{
|
|
|
- if (count >= {int(limit)}) break;
|
|
|
- const titleEl = row.querySelector(
|
|
|
- '[class*="title"], [class*="Title"], [class*="name"], [class*="content"], a[href*="/explore/"]'
|
|
|
- );
|
|
|
- const titleFull = (titleEl ? titleEl.textContent : row.childNodes[0] ? row.childNodes[0].textContent : row.textContent)
|
|
|
- .trim();
|
|
|
- if (!titleFull || titleFull.length < 2 || /(阅读|点赞|评论|收藏|发布)/.test(titleFull)) continue;
|
|
|
- // 收集该行所有数字
|
|
|
- const nums = Array.from(row.querySelectorAll('td, [class*="stat"], [class*="value"], [class*="count"]'))
|
|
|
- .map(el => el.textContent.replace(/[, ]/g, ''))
|
|
|
- .filter(t => /^\\d+$/.test(t))
|
|
|
- .map(Number);
|
|
|
- // 常见顺序: 阅读 / 点赞? / 收藏 / 评论,或不含单位。按存在数量尽力填充。
|
|
|
- const note = {{
|
|
|
- title: titleFull,
|
|
|
- publish_date: '',
|
|
|
- read_count: nums[0] || 0,
|
|
|
- like_count: nums[1] || 0,
|
|
|
- collect_count: nums[2] || 0,
|
|
|
- comment_count: nums[3] || 0,
|
|
|
- share_count: nums[4] || 0,
|
|
|
- unreplied_comment_count: 0,
|
|
|
- }};
|
|
|
- // 找发布日期
|
|
|
- const m = row.textContent.match(/(\\d{{4}}[-/]\\d{{1,2}}[-/]\\d{{1,2}})|(\\d{{1,2}}[-/]\\d{{1,2}})/);
|
|
|
- if (m) note.publish_date = m[0];
|
|
|
- notes.push(note);
|
|
|
- count++;
|
|
|
- }}
|
|
|
- return JSON.stringify(notes);
|
|
|
- }}"""
|
|
|
+ # --- 近30日:点击 tab 抓第二组 ---
|
|
|
+ period_30d = {}
|
|
|
try:
|
|
|
- raw = publisher._evaluate(expr)
|
|
|
- if isinstance(raw, str):
|
|
|
- parsed = json.loads(raw) if raw.strip().startswith("[") else []
|
|
|
- for n in parsed:
|
|
|
- n["title"] = _clean_title(n.get("title", ""))
|
|
|
- return parsed
|
|
|
+ js = (
|
|
|
+ "(() => {"
|
|
|
+ " const els = [...document.querySelectorAll('*')] ;"
|
|
|
+ " const t = els.find(e => e.children.length === 0 && e.textContent.trim() === '近30日');"
|
|
|
+ " if (t) { t.click(); return 'CLICKED'; }"
|
|
|
+ " return 'NOT_FOUND';"
|
|
|
+ "})()"
|
|
|
+ )
|
|
|
+ r = publisher._evaluate(js)
|
|
|
+ if r == "CLICKED":
|
|
|
+ time.sleep(5)
|
|
|
+ text2 = publisher._evaluate("document.body.innerText") or ""
|
|
|
+ period_30d = _parse_period(text2)
|
|
|
+ _log(f" 近30日: 观看={period_30d.get('观看数')}")
|
|
|
+ else:
|
|
|
+ _log(f" [WARN] 切换近30日 tab 失败: {r}")
|
|
|
except Exception as e:
|
|
|
- _log(f"[warn] DOM 提取笔记失败: {e}")
|
|
|
- return []
|
|
|
+ _log(f" [WARN] 抓近30日失败: {e}")
|
|
|
+
|
|
|
+ overview["period_7d"] = period_7d
|
|
|
+ overview["period_30d"] = period_30d
|
|
|
+ return overview
|
|
|
|
|
|
|
|
|
-def collect_notes_fallback(publisher) -> list:
|
|
|
- """兜底:从页面纯文本按行解析笔记数据。"""
|
|
|
+def collect_notes(publisher, limit=10):
|
|
|
+ """
|
|
|
+ 从笔记管理页(/new/note-manager)采集笔记列表。
|
|
|
+
|
|
|
+ 该页每篇笔记在 innerText 中呈现为固定行块:
|
|
|
+ 标题
|
|
|
+ YYYY-MM-DD HH:MM
|
|
|
+ 阅读数
|
|
|
+ 点赞数
|
|
|
+ 收藏数
|
|
|
+ 评论数
|
|
|
+ 分享数
|
|
|
+ """
|
|
|
text = publisher._evaluate("document.body.innerText") or ""
|
|
|
+ lines = [ln.strip() for ln in text.split("\n")]
|
|
|
+ dates = re.compile(r"^\d{4}-\d{2}-\d{2} \d{2}:\d{2}$")
|
|
|
notes = []
|
|
|
- lines = [ln.strip() for ln in text.split("\n") if ln.strip()]
|
|
|
i = 0
|
|
|
- while i < len(lines) and len(notes) < 10:
|
|
|
- line = lines[i]
|
|
|
- # 数据行特征:含 2+ 个数字
|
|
|
- nums = re.findall(r"[\d.,万]+", line)
|
|
|
- if len(nums) >= 2 and not re.match(r'^\d+$', line):
|
|
|
- # 标题 = 上一行若非数字行
|
|
|
- title = lines[i - 1] if i > 0 and not re.fullmatch(r"[\d,\.]+", lines[i - 1]) else line
|
|
|
- values = [_to_int(nm) for nm in nums]
|
|
|
- note = {
|
|
|
- "title": _clean_title(title),
|
|
|
- "publish_date": "",
|
|
|
- "read_count": values[0],
|
|
|
- "like_count": values[1] if len(values) > 1 else 0,
|
|
|
- "collect_count": values[2] if len(values) > 2 else 0,
|
|
|
- "comment_count": values[3] if len(values) > 3 else 0,
|
|
|
- "share_count": values[4] if len(values) > 4 else 0,
|
|
|
- "unreplied_comment_count": 0,
|
|
|
- }
|
|
|
- notes.append(note)
|
|
|
- i += 1
|
|
|
- return notes
|
|
|
-
|
|
|
-
|
|
|
-def collect_notes(publisher, limit=10):
|
|
|
- """采集笔记列表:先 DOM,失败则兜底文本。"""
|
|
|
- notes = collect_notes_from_dom(publisher, limit)
|
|
|
- if not notes:
|
|
|
- notes = collect_notes_fallback(publisher)
|
|
|
+ while i < len(lines) and len(notes) < limit:
|
|
|
+ ln = lines[i]
|
|
|
+ # 日期行:上一行是标题,后续连续 5 行是数字
|
|
|
+ if dates.match(ln) and i + 5 < len(lines):
|
|
|
+ title = _clean_title(lines[i - 1]) if i >= 1 else ""
|
|
|
+ nums = []
|
|
|
+ # 收集日期行后连续的数字行
|
|
|
+ j = i + 1
|
|
|
+ while j < len(lines) and len(nums) < 5 and lines[j].isdigit():
|
|
|
+ nums.append(int(lines[j]))
|
|
|
+ j += 1
|
|
|
+ if len(nums) >= 5 and title:
|
|
|
+ notes.append({
|
|
|
+ "title": title,
|
|
|
+ "publish_date": ln[:10],
|
|
|
+ "read_count": nums[0],
|
|
|
+ "like_count": nums[1],
|
|
|
+ "collect_count": nums[2],
|
|
|
+ "comment_count": nums[3],
|
|
|
+ "share_count": nums[4],
|
|
|
+ "unreplied_comment_count": 0,
|
|
|
+ })
|
|
|
+ i = j if j > i + 1 else i + 1
|
|
|
+ else:
|
|
|
+ i += 1
|
|
|
return notes[:limit]
|
|
|
|
|
|
|
|
|
-def collect_unreplied_comments(publisher) -> int:
|
|
|
- """采集未回复评论数(跳转评论消息页)。"""
|
|
|
- publisher._navigate(COMMENTS_URL)
|
|
|
- time.sleep(2)
|
|
|
- try:
|
|
|
- text = publisher._evaluate("document.body.innerText") or ""
|
|
|
- except Exception:
|
|
|
- return 0
|
|
|
- m = re.search(r"未回复[^\\d]*(\\d+)", text)
|
|
|
- if m:
|
|
|
- return int(m.group(1))
|
|
|
- # 备选:查未读提醒徽标 / 未回复 tab 计数
|
|
|
- m2 = re.search(r"未回复评论[^\\d]*(\\d+)", text)
|
|
|
- return int(m2.group(1)) if m2 else 0
|
|
|
-
|
|
|
-
|
|
|
def screenshot_current(publisher, output_path):
|
|
|
"""用 CDP 截取当前页面(JPEG)。"""
|
|
|
os.makedirs(os.path.dirname(output_path), exist_ok=True)
|
|
|
result = publisher._send("Page.captureScreenshot", {"format": "jpeg", "quality": 60})
|
|
|
- if result.get("result", {}).get("data"):
|
|
|
- img_data = base64.b64decode(result["result"]["data"])
|
|
|
+ # CDP 对 captureScreenshot 返回 {"data": "<base64>"}(_send 已剥掉 result 包装)
|
|
|
+ data = result.get("data") or (result.get("result") or {}).get("data")
|
|
|
+ if data:
|
|
|
+ img_data = base64.b64decode(data)
|
|
|
with open(output_path, "wb") as f:
|
|
|
f.write(img_data)
|
|
|
_log(f"截图已保存: {output_path} ({len(img_data)} bytes)")
|
|
|
@@ -291,22 +278,16 @@ def main():
|
|
|
_log(f" 粉丝数: {overview.get('fans_count')}, 总阅读: {overview.get('total_read')}, "
|
|
|
f"总互动: {overview.get('total_interact')}, 笔记数: {overview.get('note_count')}")
|
|
|
|
|
|
- # 2. 笔记列表
|
|
|
_log(f"采集笔记列表(最多 {args.limit} 条)...")
|
|
|
publisher._navigate(NOTES_MANAGE_URL)
|
|
|
- time.sleep(3)
|
|
|
+ time.sleep(4)
|
|
|
notes = collect_notes(publisher, args.limit)
|
|
|
_log(f" 采集到 {len(notes)} 篇笔记")
|
|
|
|
|
|
- # 3. 未回复评论
|
|
|
- _log("采集未回复评论数...")
|
|
|
- unreplied = collect_unreplied_comments(publisher)
|
|
|
- for n in notes:
|
|
|
- n["unreplied_comment_count"] = unreplied
|
|
|
- _log(f" 未回复评论: {unreplied}")
|
|
|
+ # 说明:新版创作者中心已无汇总"未回复评论"独立页面,评论数据
|
|
|
+ # 体现在每篇笔记的 comment_count 中;如需每篇未回复详情需进单篇详情页。
|
|
|
|
|
|
- # 4. 截图(回到创作者首页,截账号概览看板)
|
|
|
- _log("截图...")
|
|
|
+ _log("截图(笔记管理页)...")
|
|
|
shot_path = os.path.join(SCREENSHOT_DIR, f"xhs_daily_{date}.jpg")
|
|
|
screenshot_current(publisher, shot_path)
|
|
|
|