Browse Source

内容: 修复数据采集近7日/近30日双周期 + 截图上传修复

Sisyphus Agent 1 month ago
parent
commit
83cfd38749
2 changed files with 142 additions and 137 deletions
  1. 38 14
      运营文案/xhs_feishu.py
  2. 104 123
      运营文案/小红书发布/_过程脚本/xhs_daily_report.py

+ 38 - 14
运营文案/xhs_feishu.py

@@ -57,10 +57,15 @@ def _get_token(app_id: str, app_secret: str) -> str:
 
 
 def _upload_image(token: str, image_path: str) -> str:
-    """上传图片到飞书,返回 image_key。"""
+    """上传图片到飞书,返回 image_key。
+
+    注意:`image_type` 是 multipart form-data 的字段(不是 URL query 参数)。
+    此前误写成 `?image_type=message` 会导致 400 Bad Request(code=234001)。
+    """
+    url = FEISHU_IMG_URL
     with open(image_path, "rb") as f:
         resp = requests.post(
-            FEISHU_IMG_URL,
+            url,
             headers={"Authorization": f"Bearer {token}"},
             files={"image": (os.path.basename(image_path), f, "image/jpeg")},
             data={"image_type": "message"},
@@ -76,19 +81,33 @@ def _upload_image(token: str, image_path: str) -> str:
 def _build_card_json(date: str, overview: dict, notes: list) -> str:
     """构建飞书富文本卡片 JSON 字符串。
 
-    overview 字段: fans_count, fans_growth_30d, total_read, total_interact, note_count
-    notes 元素字段: title, publish_date, read_count, like_count, collect_count,
-                    comment_count, share_count, unreplied_comment_count
+    overview 字段: fans_count, interact_count, follow_count,
+                   period_7d: {曝光数/观看数/点赞数/评论数/收藏数/分享数/净涨粉...},
+                   period_30d: 同上
     """
     ov = overview or {}
+
+    def _period_lines(period, label):
+        """格式化一个周期段"""
+        if not period:
+            return [f"⚠️ {label}数据暂不可用"]
+        return [
+            f"{label} 曝光 {period.get('曝光数', '-')}",
+            f"     观看 {period.get('观看数', '-')} | 点赞 {period.get('点赞数', '-')} | "
+            f"评论 {period.get('评论数', '-')} | 收藏 {period.get('收藏数', '-')} | "
+            f"分享 {period.get('分享数', '-')}",
+            f"     净涨粉 {period.get('净涨粉', '-')}",
+        ]
+
     overview_lines = [
-        f"粉丝数:{ov.get('fans_count', 0):,}"
-        f"(近30天 {'+' if ov.get('fans_growth_30d', 0) >= 0 else ''}"
-        f"{ov.get('fans_growth_30d', 0):,})",
-        f"总阅读量:{ov.get('total_read', 0):,}",
-        f"总互动量:{ov.get('total_interact', 0):,}",
-        f"笔记总数:{ov.get('note_count', 0)}",
+        f"粉丝数:{ov.get('fans_count', 0)}  关注:{ov.get('follow_count', 0)}  "
+        f"获赞与收藏:{ov.get('interact_count', 0)}",
     ]
+    # 近7日
+    overview_lines += _period_lines(ov.get("period_7d", {}), "近7日")
+    # 近30日(如果有)
+    if ov.get("period_30d"):
+        overview_lines += _period_lines(ov.get("period_30d", {}), "近30日")
     note_lines = []
     for i, n in enumerate(notes, 1):
         note_lines.append(
@@ -122,15 +141,20 @@ def _build_card_json(date: str, overview: dict, notes: list) -> str:
 
 
 def _send_message(token: str, cfg: dict, msg_type: str, content: str) -> dict:
-    """发送一条飞书消息,返回响应数据。"""
+    """发送一条飞书消息,返回响应数据。
+
+    重要:飞书 im/v1/messages 接口的 receive_id_type 是 URL 查询参数,
+    不是 JSON body 字段(否则报 99992402 receive_id_type is required)。
+    """
+    rtype = cfg.get("receive_id_type", "chat_id")
+    url = f"{FEISHU_MSG_URL}?receive_id_type={rtype}"
     body = {
-        "receive_id_type": cfg.get("receive_id_type", "chat_id"),
         "receive_id": cfg["feishu_chat_id"],
         "msg_type": msg_type,
         "content": content,
     }
     resp = requests.post(
-        FEISHU_MSG_URL,
+        url,
         headers={"Authorization": f"Bearer {token}"},
         json=body,
         timeout=30,

+ 104 - 123
运营文案/小红书发布/_过程脚本/xhs_daily_report.py

@@ -47,8 +47,7 @@ XHS_SKILLS_PATH = r"C:\code\XiaohongshuSkills\scripts"
 sys.path.insert(0, XHS_SKILLS_PATH)
 
 CREATOR_HOME = "https://creator.xiaohongshu.com/new/home"
-NOTES_MANAGE_URL = "https://creator.xiaohongshu.com/publish/publish_manage"
-COMMENTS_URL = "https://creator.xiaohongshu.com/message/comment"
+NOTES_MANAGE_URL = "https://creator.xiaohongshu.com/new/note-manager"
 
 
 def _log(msg):
@@ -97,137 +96,125 @@ def _clean_title(title):
 
 
 def collect_overview(publisher) -> dict:
-    """从创作者中心首页获取账号总览数据。"""
+    """
+    从创作者中心首页 /new/home 获取账号总览数据。
+
+    页面 innerText 结构(实测):
+      - 账号级字段: "数字\\n标签"(如 "20\\n粉丝数", "551\\n获赞与收藏")
+      - 周期数据: "标签\\n数字\\n环比xx%"(如 "观看数\\n209\\n环比-86%")
+      - 默认显示近7日,点击「近30日」tab 抓第二组
+
+    返回 dict:
+      fans_count / follow_count / interact_count(账号级)
+      period_7d / period_30d: 各含 exposure/view/like/comment/collect/share/net_fans
+    """
     overview = {}
-    text = (publisher._evaluate("document.body.innerText") or "")
 
-    # 粉丝数 / 总阅读 / 总互动 / 笔记总数(宽泛匹配,兼容页面 word 差异)
-    patterns = [
-        ("fans_count", r"粉丝数\s*[::\n]\s*([\d.,万]+)"),
-        ("total_read", r"总阅读[量]?\s*[::\n]\s*([\d.,万]+)"),
-        ("total_interact", r"总互动[量]?\s*[::\n]\s*([\d.,万]+)"),
-        ("note_count", r"笔记总数\s*[::\n]\s*([\d.,]+)"),
-        ("fans_growth_30d", r"(?:近30天|近3[0-9]天)[^\d\n]*?([+-]?[\d.,万]+)"),
-    ]
-    for key, pat in patterns:
-        m = re.search(pat, text)
+    def _parse_period(text):
+        """解析周期数据(标签\\n数字\\n环比%)。返回 dict[label] = (值, 原始字符串)"""
+        p = {}
+        # 匹配 "标签\n数字\n环比xx%" 或 "标签\n数字\n环比-"
+        pattern = re.compile(
+            r"^(曝光数|观看数|封面点击率|视频完播率|点赞数|评论数|收藏数|分享数|净涨粉|新增关注|取消关注|主页访客)"
+            r"\n([\d.,万%]+)\n环比([+-]?\d*%?)", re.M)
+        for m in pattern.finditer(text):
+            p[m.group(1)] = m.group(2)
+        return p
+
+    # --- 账号级字段:数字在标签前 ---
+    text = (publisher._evaluate("document.body.innerText") or "")
+    for key, label in (("fans_count", "粉丝数"), ("follow_count", "关注数"),
+                       ("interact_count", "获赞与收藏")):
+        m = re.search(r"([\d.,万]+)\n" + re.escape(label), text)
         if m:
             overview[key] = _to_int(m.group(1))
-    return overview
+        else:
+            overview[key] = 0
 
+    # --- 近7日(默认 tab) ---
+    period_7d = _parse_period(text)
 
-def collect_notes_from_dom(publisher, limit=10):
-    """从笔记管理页 DOM 提取笔记列表(完整标题 + 数据)。"""
-    expr = f"""() => {{
-        const notes = [];
-        // 常见容器选择器:数据表格行 / 笔记卡片
-        const rows = document.querySelectorAll(
-            'table tbody tr, .note-item, [class*="note-row"], [class*="note-item"], [class*="data-card"], li'
-        );
-        let count = 0;
-        for (const row of rows) {{
-            if (count >= {int(limit)}) break;
-            const titleEl = row.querySelector(
-                '[class*="title"], [class*="Title"], [class*="name"], [class*="content"], a[href*="/explore/"]'
-            );
-            const titleFull = (titleEl ? titleEl.textContent : row.childNodes[0] ? row.childNodes[0].textContent : row.textContent)
-                .trim();
-            if (!titleFull || titleFull.length < 2 || /(阅读|点赞|评论|收藏|发布)/.test(titleFull)) continue;
-            // 收集该行所有数字
-            const nums = Array.from(row.querySelectorAll('td, [class*="stat"], [class*="value"], [class*="count"]'))
-                .map(el => el.textContent.replace(/[, ]/g, ''))
-                .filter(t => /^\\d+$/.test(t))
-                .map(Number);
-            // 常见顺序: 阅读 / 点赞? / 收藏 / 评论,或不含单位。按存在数量尽力填充。
-            const note = {{
-                title: titleFull,
-                publish_date: '',
-                read_count: nums[0] || 0,
-                like_count: nums[1] || 0,
-                collect_count: nums[2] || 0,
-                comment_count: nums[3] || 0,
-                share_count: nums[4] || 0,
-                unreplied_comment_count: 0,
-            }};
-            // 找发布日期
-            const m = row.textContent.match(/(\\d{{4}}[-/]\\d{{1,2}}[-/]\\d{{1,2}})|(\\d{{1,2}}[-/]\\d{{1,2}})/);
-            if (m) note.publish_date = m[0];
-            notes.push(note);
-            count++;
-        }}
-        return JSON.stringify(notes);
-    }}"""
+    # --- 近30日:点击 tab 抓第二组 ---
+    period_30d = {}
     try:
-        raw = publisher._evaluate(expr)
-        if isinstance(raw, str):
-            parsed = json.loads(raw) if raw.strip().startswith("[") else []
-            for n in parsed:
-                n["title"] = _clean_title(n.get("title", ""))
-            return parsed
+        js = (
+            "(() => {"
+            "  const els = [...document.querySelectorAll('*')] ;"
+            "  const t = els.find(e => e.children.length === 0 && e.textContent.trim() === '近30日');"
+            "  if (t) { t.click(); return 'CLICKED'; }"
+            "  return 'NOT_FOUND';"
+            "})()"
+        )
+        r = publisher._evaluate(js)
+        if r == "CLICKED":
+            time.sleep(5)
+            text2 = publisher._evaluate("document.body.innerText") or ""
+            period_30d = _parse_period(text2)
+            _log(f"  近30日: 观看={period_30d.get('观看数')}")
+        else:
+            _log(f"  [WARN] 切换近30日 tab 失败: {r}")
     except Exception as e:
-        _log(f"[warn] DOM 提取笔记失败: {e}")
-    return []
+        _log(f"  [WARN] 抓近30日失败: {e}")
+
+    overview["period_7d"] = period_7d
+    overview["period_30d"] = period_30d
+    return overview
 
 
-def collect_notes_fallback(publisher) -> list:
-    """兜底:从页面纯文本按行解析笔记数据。"""
+def collect_notes(publisher, limit=10):
+    """
+    从笔记管理页(/new/note-manager)采集笔记列表。
+
+    该页每篇笔记在 innerText 中呈现为固定行块:
+        标题
+        YYYY-MM-DD HH:MM
+        阅读数
+        点赞数
+        收藏数
+        评论数
+        分享数
+    """
     text = publisher._evaluate("document.body.innerText") or ""
+    lines = [ln.strip() for ln in text.split("\n")]
+    dates = re.compile(r"^\d{4}-\d{2}-\d{2} \d{2}:\d{2}$")
     notes = []
-    lines = [ln.strip() for ln in text.split("\n") if ln.strip()]
     i = 0
-    while i < len(lines) and len(notes) < 10:
-        line = lines[i]
-        # 数据行特征:含 2+ 个数字
-        nums = re.findall(r"[\d.,万]+", line)
-        if len(nums) >= 2 and not re.match(r'^\d+$', line):
-            # 标题 = 上一行若非数字行
-            title = lines[i - 1] if i > 0 and not re.fullmatch(r"[\d,\.]+", lines[i - 1]) else line
-            values = [_to_int(nm) for nm in nums]
-            note = {
-                "title": _clean_title(title),
-                "publish_date": "",
-                "read_count": values[0],
-                "like_count": values[1] if len(values) > 1 else 0,
-                "collect_count": values[2] if len(values) > 2 else 0,
-                "comment_count": values[3] if len(values) > 3 else 0,
-                "share_count": values[4] if len(values) > 4 else 0,
-                "unreplied_comment_count": 0,
-            }
-            notes.append(note)
-        i += 1
-    return notes
-
-
-def collect_notes(publisher, limit=10):
-    """采集笔记列表:先 DOM,失败则兜底文本。"""
-    notes = collect_notes_from_dom(publisher, limit)
-    if not notes:
-        notes = collect_notes_fallback(publisher)
+    while i < len(lines) and len(notes) < limit:
+        ln = lines[i]
+        # 日期行:上一行是标题,后续连续 5 行是数字
+        if dates.match(ln) and i + 5 < len(lines):
+            title = _clean_title(lines[i - 1]) if i >= 1 else ""
+            nums = []
+            # 收集日期行后连续的数字行
+            j = i + 1
+            while j < len(lines) and len(nums) < 5 and lines[j].isdigit():
+                nums.append(int(lines[j]))
+                j += 1
+            if len(nums) >= 5 and title:
+                notes.append({
+                    "title": title,
+                    "publish_date": ln[:10],
+                    "read_count": nums[0],
+                    "like_count": nums[1],
+                    "collect_count": nums[2],
+                    "comment_count": nums[3],
+                    "share_count": nums[4],
+                    "unreplied_comment_count": 0,
+                })
+            i = j if j > i + 1 else i + 1
+        else:
+            i += 1
     return notes[:limit]
 
 
-def collect_unreplied_comments(publisher) -> int:
-    """采集未回复评论数(跳转评论消息页)。"""
-    publisher._navigate(COMMENTS_URL)
-    time.sleep(2)
-    try:
-        text = publisher._evaluate("document.body.innerText") or ""
-    except Exception:
-        return 0
-    m = re.search(r"未回复[^\\d]*(\\d+)", text)
-    if m:
-        return int(m.group(1))
-    # 备选:查未读提醒徽标 / 未回复 tab 计数
-    m2 = re.search(r"未回复评论[^\\d]*(\\d+)", text)
-    return int(m2.group(1)) if m2 else 0
-
-
 def screenshot_current(publisher, output_path):
     """用 CDP 截取当前页面(JPEG)。"""
     os.makedirs(os.path.dirname(output_path), exist_ok=True)
     result = publisher._send("Page.captureScreenshot", {"format": "jpeg", "quality": 60})
-    if result.get("result", {}).get("data"):
-        img_data = base64.b64decode(result["result"]["data"])
+    # CDP 对 captureScreenshot 返回 {"data": "<base64>"}(_send 已剥掉 result 包装)
+    data = result.get("data") or (result.get("result") or {}).get("data")
+    if data:
+        img_data = base64.b64decode(data)
         with open(output_path, "wb") as f:
             f.write(img_data)
         _log(f"截图已保存: {output_path} ({len(img_data)} bytes)")
@@ -291,22 +278,16 @@ def main():
         _log(f"  粉丝数: {overview.get('fans_count')}, 总阅读: {overview.get('total_read')}, "
              f"总互动: {overview.get('total_interact')}, 笔记数: {overview.get('note_count')}")
 
-        # 2. 笔记列表
         _log(f"采集笔记列表(最多 {args.limit} 条)...")
         publisher._navigate(NOTES_MANAGE_URL)
-        time.sleep(3)
+        time.sleep(4)
         notes = collect_notes(publisher, args.limit)
         _log(f"  采集到 {len(notes)} 篇笔记")
 
-        # 3. 未回复评论
-        _log("采集未回复评论数...")
-        unreplied = collect_unreplied_comments(publisher)
-        for n in notes:
-            n["unreplied_comment_count"] = unreplied
-        _log(f"  未回复评论: {unreplied}")
+        # 说明:新版创作者中心已无汇总"未回复评论"独立页面,评论数据
+        # 体现在每篇笔记的 comment_count 中;如需每篇未回复详情需进单篇详情页。
 
-        # 4. 截图(回到创作者首页,截账号概览看板)
-        _log("截图...")
+        _log("截图(笔记管理页)...")
         shot_path = os.path.join(SCREENSHOT_DIR, f"xhs_daily_{date}.jpg")
         screenshot_current(publisher, shot_path)