Browse Source

feat(docs): add dan_reports - assessment reports and processing scripts

User 2 months ago
parent
commit
4c5c1325a2

+ 134 - 0
docs/参考资料/dan_reports/dedup_reports.py

@@ -0,0 +1,134 @@
+# -*- coding: utf-8 -*-
+"""
+去重重复报告:找出 姓名+生日+测评日期+报告类型 完全相同的记录,
+保留格式规范的最新文件,删除旧格式的重复PDF。
+
+用法: python scripts/reports/dedup_reports.py
+"""
+import os, csv, re, shutil
+from collections import Counter
+
+PROJECT_DIR = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
+REPORT_DIR = os.path.join(PROJECT_DIR, '报告')
+CSV_PATH = os.path.join(PROJECT_DIR, '素材库', '测评数据_优化提取.csv')
+BACKUP_PATH = CSV_PATH.replace('.csv', '_before_dedup.csv')
+
+
+def is_proper_format(filename):
+    """判断是否为规范格式的文件名(保留此类)。
+    
+    规范格式:
+      - 带时间戳: "B4 核心认知能力和成长型思维_姓名_202403231742254603.pdf"
+      - 无时间戳: "B6_生涯规划(高中版)_姓名.pdf"
+      - 核心特征: 文件名不包含 "DAN素质能力测评报告" 等旧格式标记,
+                 且前缀类型与文件名其他内容之间不含旧命名痕迹
+    
+    旧格式: "A1_DAN素质能力测评报告_姓名*.pdf", 
+            "A2_B6生涯规划(高中版)_姓名*.pdf",
+            "B6_DAN素质能力测评报告-姓名*.pdf"
+    """
+    # 旧格式特征:包含 DAN素质能力测评报告 字样
+    if 'DAN素质能力测评报告' in filename:
+        return False
+    
+    # 旧格式特征:A2_B6 开头的旧生涯规划文件
+    if filename.startswith('A2_B6'):
+        return False
+    
+    # 规范格式特征:
+    # 1. {类型}{空格}{中文描述}_{姓名}_{时间戳(14-18位)}.pdf
+    if re.match(r'^[A-Z]\d\s+\S+?_\S+?_\d{14,}\.pdf$', filename):
+        return True
+    # 2. {类型}_{中文描述}_{姓名}.pdf(无时间戳,有中文描述字符≥3)
+    if re.match(r'^[A-Z]\d_\S{3,}_.+\.pdf$', filename):
+        return True
+    
+    return False
+
+
+def main():
+    # 1. 读取CSV
+    with open(CSV_PATH, 'r', encoding='utf-8-sig') as f:
+        reader = csv.DictReader(f)
+        rows = list(reader)
+        fieldnames = reader.fieldnames
+    
+    print(f'总记录数: {len(rows)}')
+    
+    # 2. 按 姓名+生日+测评日期+报告类型 分组
+    groups = {}
+    for r in rows:
+        k = (r.get('姓名', ''), r.get('生日', ''), r.get('测评日期', ''), r.get('报告类型', ''))
+        groups.setdefault(k, []).append(r)
+    
+    # 3. 找出重复组
+    dup_groups = {k: v for k, v in groups.items() if len(v) > 1}
+    print(f'重复组数: {len(dup_groups)}')
+    
+    to_delete_files = []
+    kept_rows = []
+    
+    for k, dup_rows in sorted(dup_groups.items(), key=lambda x: -len(x[1])):
+        name, birthday, date, rtype = k
+        print(f'\n组: {name} | {birthday} | {date} | {rtype} (x{len(dup_rows)})')
+        
+        # 按文件规范程度排序:规范格式优先保留
+        # 如果多个规范格式,保留文件名排序最后的(时间戳最新的)
+        proper = [r for r in dup_rows if is_proper_format(r.get('filename', ''))]
+        old = [r for r in dup_rows if not is_proper_format(r.get('filename', ''))]
+        
+        if proper:
+            # 保留规范格式中时间戳最新的
+            proper.sort(key=lambda r: r.get('filename', ''), reverse=True)
+            keep = proper[0]
+            delete = proper[1:] + old
+        else:
+            # 没有规范格式,保留文件名排序最后的
+            dup_rows.sort(key=lambda r: r.get('filename', ''), reverse=True)
+            keep = dup_rows[0]
+            delete = dup_rows[1:]
+        
+        print(f'  保留: {keep["filename"][:50]}')
+        kept_rows.append(keep)
+        
+        for r in delete:
+            fname = r['filename']
+            fpath = os.path.join(REPORT_DIR, fname)
+            print(f'  删除: {fname[:50]}')
+            if os.path.exists(fpath):
+                to_delete_files.append(fpath)
+    
+    # 4. 加入非重复的记录
+    non_dup = [rows for k, rows in groups.items() if len(rows) == 1]
+    for group in non_dup:
+        kept_rows.extend(group)
+    
+    print(f'\n保留记录: {len(kept_rows)}')
+    print(f'待删除文件: {len(to_delete_files)}')
+    
+    # 5. 备份原CSV
+    shutil.copy2(CSV_PATH, BACKUP_PATH)
+    print(f'备份CSV: {BACKUP_PATH}')
+    
+    # 6. 写入新CSV
+    with open(CSV_PATH, 'w', encoding='utf-8-sig', newline='') as f:
+        writer = csv.DictWriter(f, fieldnames=fieldnames)
+        writer.writeheader()
+        writer.writerows(kept_rows)
+    print(f'写入新CSV: {CSV_PATH} ({len(kept_rows)} 记录)')
+    
+    # 7. 删除旧文件
+    if to_delete_files:
+        print(f'\n删除 {len(to_delete_files)} 个重复PDF...')
+        for fpath in to_delete_files:
+            try:
+                os.remove(fpath)
+                print(f'  DELETED: {os.path.basename(fpath)}')
+            except Exception as e:
+                print(f'  FAILED: {os.path.basename(fpath)} - {e}')
+    
+    print(f'\n完成!共去重 {sum(len(v) for v in dup_groups.values()) - len(dup_groups)} 条记录,删除 {len(to_delete_files)} 个文件')
+
+
+if __name__ == '__main__':
+    main()

+ 64 - 0
docs/参考资料/dan_reports/delete_duplicate_a1.py

@@ -0,0 +1,64 @@
+#!/usr/bin/env python
+# -*- coding: utf-8 -*-
+"""
+检查报告中所有A1_开头的文件,如果同目录中有内容完全相同的文件,则删掉这个A1_开头的文件
+"""
+
+import os
+import hashlib
+from pathlib import Path
+
+def file_hash(filepath):
+    """计算文件 SHA256 哈希"""
+    h = hashlib.sha256()
+    with open(filepath, 'rb') as f:
+        for chunk in iter(lambda: f.read(65536), b''):
+            h.update(chunk)
+    return h.hexdigest()
+
+def main():
+    # 自动计算项目根目录 (scripts/reports/ → 项目根)
+    report_dir = Path(__file__).parent.parent.parent / "报告"
+    
+    # 找出所有 A1_ 开头的文件
+    a1_files = list(report_dir.glob("A1_*.pdf"))
+    print(f"找到 {len(a1_files)} 个 A1_ 开头的文件")
+    
+    # 计算所有文件的哈希
+    all_files = list(report_dir.glob("*.pdf"))
+    hash_map = {}
+    
+    for f in all_files:
+        h = file_hash(str(f))
+        if h not in hash_map:
+            hash_map[h] = []
+        hash_map[h].append(f)
+    
+    # 找出重复的文件
+    deleted_count = 0
+    for h, files in hash_map.items():
+        if len(files) > 1:
+            a1_dups = [f for f in files if f.name.startswith("A1_")]
+            non_a1 = [f for f in files if not f.name.startswith("A1_")]
+            
+            if a1_dups and non_a1:
+                for f in a1_dups:
+                    try:
+                        f.unlink()
+                        print(f"[DELETED] {f.name} (与 {non_a1[0].name} 内容相同)")
+                        deleted_count += 1
+                    except Exception as e:
+                        print(f"[FAIL] {f.name}: {e}")
+            elif len(a1_dups) > 1:
+                keep = a1_dups[0]
+                for f in a1_dups[1:]:
+                    try:
+                        f.unlink()
+                        print(f"[DELETED] {f.name} (与 {keep.name} 内容相同)")
+                        deleted_count += 1
+                    except Exception as e:
+                        print(f"[FAIL] {f.name}: {e}")
+    print(f"\n总共删除了 {deleted_count} 个重复的 A1_ 文件")
+
+if __name__ == "__main__":
+    main()

+ 1266 - 0
docs/参考资料/dan_reports/extract_all_types.py

@@ -0,0 +1,1266 @@
+# -*- coding: utf-8 -*-
+"""
+完善素材库 - 处理所有报告类型
+提取A1, A2, B2, B3, B4, B6, C1报告数据
+"""
+import fitz
+import re
+import os
+import csv
+from type_detector import detect_type
+import math
+import io
+from PIL import Image
+
+def extract_text_from_pdf(pdf_path, return_pages=False):
+    """使用PyMuPDF提取文本"""
+    try:
+        doc = fitz.open(pdf_path)
+        
+        if return_pages:
+            page_texts = []
+            for page in doc:
+                page_texts.append(page.get_text())
+            doc.close()
+            return page_texts
+        
+        text = ""
+        for page in doc:
+            text += page.get_text()
+        doc.close()
+        return text
+    except Exception as e:
+        print(f"Error extracting text from {pdf_path}: {e}")
+        return "" if not return_pages else []
+
+def extract_name_from_filename(filename):
+    """从文件名提取姓名 - 改进版"""
+    name = filename.replace('.pdf', '')
+    
+    # 优先尝试从文件名中提取姓名
+    # 注意:只用下划线分割,不用连字符(-),避免把"(1-3年级)"切碎
+    parts = re.split(r'_', name)
+    
+    # 先找包含中文的部分(姓名通常含中文)
+    chinese_parts = []
+    for p in parts:
+        if re.search(r'[\u4e00-\u9fff]{2,}', p):
+            # 排除包含关键词的
+            keywords = ['儿童', '核心', '认知', '发展', 'DAN', '测评', '自我', '综合', 
+                       '家庭', '职业', '校园', '准备', '素养', '人格', '思维', '能力',
+                       '青少年', '素质', '成长', '学习', '动机', '兴趣', '规划',
+                       '认知能力', '成长型', '多元', '智能', '教养', '亲子', '欺凌',
+                       '心理', '教育', '教师', '反馈', '报告', '班级', '校级',
+                       '加工速度', '注意力', '记忆力', '推理', '空间', '知觉',
+                       'B5', 'B6', 'C1', '标准版', '专业版', '高中段']
+            p_clean = re.sub(r'[A-Za-z0-9\-()]', '', p)  # 去掉字母数字和连字符/括号
+            if p_clean and not any(kw in p for kw in keywords):
+                chinese_parts.append(p_clean)
+    
+    if chinese_parts:
+        # 取第一个/最长的中文名;等长时优先取不含特殊字符的
+        max_len = max(len(c) for c in chinese_parts)
+        candidates = [c for c in chinese_parts if len(c) == max_len]
+        # 等长时取最不像描述文本的(包含 级/挑战/版/段 这类字眼的靠后)
+        if len(candidates) > 1:
+            bad_markers = ['级', '挑战', '版', '段', '测', '试', '报', '告']
+            def badness(s):
+                return sum(1 for ch in bad_markers if ch in s)
+            candidates.sort(key=lambda s: (badness(s), chinese_parts.index(s)))
+        return candidates[0]
+    
+    # 备选1: 对含连字符(-)的part再按-拆分(兼容旧版DAN格式如"素质能力测评-小明2024-10-069193")
+    sub_chinese = []
+    for p in parts:
+        if '-' not in p:
+            continue
+        sub_parts = re.split(r'-', p)
+        for sp in sub_parts:
+            if re.search(r'[\u4e00-\u9fff]{2,}', sp):
+                sp_clean = re.sub(r'[A-Za-z0-9\-()]', '', sp)
+                if sp_clean and not any(kw in sp for kw in keywords):
+                    sub_chinese.append(sp_clean)
+    if sub_chinese:
+        max_len = max(len(c) for c in sub_chinese)
+        candidates = [c for c in sub_chinese if len(c) == max_len]
+        if len(candidates) > 1:
+            bad_markers = ['级', '挑战', '版', '段', '测', '试', '报', '告']
+            candidates.sort(key=lambda s: (sum(1 for ch in bad_markers if ch in s), sub_chinese.index(s)))
+        return candidates[0]
+    
+    # 备选2: 按关键词过滤整块part
+    keywords = ['儿童', '核心', '认知', '发展', 'A1', 'A2', 'B2', 'B3', 'B4', 'B5', 'B6', 
+                'DAN', '测评', '自我', '综合', '家庭', '职业', '校园', '准备', '素养', 
+                '人格', '思维', '能力', '青少年', '素质', '成长', '学习', '动机', '兴趣',
+                '规划', '加工速度', '注意力', '记忆力', '推理', '空间', '知觉']
+    for p in parts:
+        if not p:
+            continue
+        if re.match(r'^\d{8,}$', p):
+            continue
+        if any(kw == p or (p.startswith(kw) and len(p) > len(kw)) for kw in keywords):
+            continue
+        if any(kw in p for kw in keywords):
+            continue
+        if len(p) >= 2 and not re.match(r'^\d+$', p):
+            return p.strip()
+    return parts[0] if parts else name
+
+def extract_birthday_from_pdf(text, page_texts=None):
+    """从PDF中提取生日"""
+    if page_texts and len(page_texts) > 1:
+        page2_text = page_texts[1]
+    else:
+        page2_text = text
+    
+    patterns = [
+        r'出生日期[::\s]*(\d{4}[-/年]\d{1,2}[-/月]\d{1,2})',
+        r'(\d{4}[-/年]\d{1,2}[-/月]\d{1,2})',
+    ]
+    
+    for pattern in patterns:
+        match = re.search(pattern, page2_text)
+        if match:
+            date_str = match.group(1)
+            date_str = date_str.replace('年', '-').replace('月', '-').replace('/', '-')
+            if re.match(r'\d{4}-\d{2}-\d{2}', date_str):
+                return date_str
+    return ""
+
+def extract_score_and_percentile(text):
+    """通用提取总分和百分位(所有报告类型共用)
+    
+    PDF格式: "112\n总得分\nTotal Score\n79\n百分位(%)"
+    总分 = "总得分"前面的数字, 百分位 = "总得分"和"百分位"之间的数字
+    """
+    result = {}
+    
+    # 主格式: 数字\n总得分  (所有类型报告通用)
+    score_match = re.search(r'(\d+)\s*\n\s*总得分', text)
+    if score_match:
+        result['总分'] = score_match.group(1)
+    
+    # 百分位: 总得分...数字...百分位
+    pct_match = re.search(r'总得分.*?(\d+)\s*百分位', text, re.DOTALL)
+    if pct_match:
+        result['百分位'] = pct_match.group(1)
+    
+    # 后备1: 旧版 "总分" 格式  
+    if '总分' not in result:
+        total_match = re.search(r'总分[^\d]*(\d+)', text)
+        if total_match:
+            result['总分'] = total_match.group(1)
+    
+    # 后备2: 旧版 "百分位" 格式
+    if '百分位' not in result:
+        pct_match2 = re.search(r'百分位[^\d]*(\d+)', text)
+        if pct_match2:
+            result['百分位'] = pct_match2.group(1)
+    
+    return result
+
+def extract_a1_data(text, page_texts=None):
+    """提取A1报告数据 - 儿童核心认知发展(6个认知维度)"""
+    result = {}
+    
+    result.update(extract_score_and_percentile(text))
+    
+    # 6个核心认知维度: 感知觉, 注意力, 记忆力, 推理能力, 空间能力, 加工速度
+    cognitive_dims = ['感知觉', '注意力', '记忆力', '推理能力', '空间能力', '加工速度']
+    
+    # 方法1: 从summary页面提取百分位 (格式: "感知觉 |  Perception\n百分位(%)\n14")
+    for dim in cognitive_dims:
+        pct_match = re.search(
+            rf'{re.escape(dim)}\s*\|\s*\w+\s*\n\s*百分位(%)\s*\n\s*(\d+)',
+            text
+        )
+        if pct_match:
+            result[f'{dim}_pct'] = pct_match.group(1)
+    
+    # 方法2: 从detail页面提取原始分和百分位 (格式: "我的感知觉得分 41  | 14%")
+    for dim in cognitive_dims:
+        detail_match = re.search(
+            rf'我的{re.escape(dim)}得分\s*(\d+)\s*\|\s*(\d+)%',
+            text
+        )
+        if detail_match:
+            result[f'{dim}_score'] = detail_match.group(1)
+            if f'{dim}_pct' not in result:
+                result[f'{dim}_pct'] = detail_match.group(2)
+    
+    # 方法3: 后备 - 直接搜索 "维度名\n百分位(%)\n数字" (某些PDF格式)
+    if len([k for k in result if k.endswith('_pct')]) < 6:
+        for dim in cognitive_dims:
+            if f'{dim}_pct' not in result:
+                fallback = re.search(
+                    rf'{re.escape(dim)}.*?百分位(%).*?(\d+)',
+                    text,
+                    re.DOTALL
+                )
+                if fallback:
+                    result[f'{dim}_pct'] = fallback.group(1)
+    
+    return result
+
+def extract_a2_data(text, page_texts=None):
+    """提取A2报告数据 - 核心素养16+(认知+人格+情绪+关系+健康)"""
+    result = {}
+    
+    # 总分和百分位 (通用格式)
+    result.update(extract_score_and_percentile(text))
+    
+    # ===== 认知维度 (3个: 感知觉, 记忆力, 注意力) =====
+    # 在summary页面中: "感知觉\n89%"
+    for dim in ['感知觉', '记忆力', '注意力']:
+        cog_match = re.search(rf'{re.escape(dim)}\s*\n\s*(\d+)%', text)
+        if cog_match:
+            result[f'{dim}_pct'] = cog_match.group(1)
+    
+    # ===== 情绪状态 (第5页/索引4) =====
+    if page_texts and len(page_texts) > 4:
+        page5_text = page_texts[4]
+        
+        # 方法1: 原始格式 "数字\nInferiority"
+        emotion_names = ['自卑-自信', '抑郁-安详', '焦虑-安详', '无力感-掌控感']
+        emotion_eng = ['Inferiority', 'Depression', 'Anxiety', 'Helpless']
+        
+        for eng, cn in zip(emotion_eng, emotion_names):
+            em_match = re.search(rf'(\d+(?:\.?\d+)?)\s*\n\s*{re.escape(eng)}', page5_text)
+            if em_match:
+                result[cn] = em_match.group(1)
+        
+        # 方法2: 旧格式 (后备)
+        if not any(k in result for k in emotion_names):
+            emotion_pattern = re.findall(r'(\d+)\s+10\s+(\d+\.?\d*)\s+(\d+\.?\d*)', page5_text)
+            if len(emotion_pattern) >= 1:
+                for i in range(min(len(emotion_pattern), 4)):
+                    result[emotion_names[i]] = emotion_pattern[i][-1]
+        
+        # 情绪总分: "在4个分测验中的总得分是 XX分"
+        total_em = re.search(r'在4个分测验中的总得分是\s*(\d+(?:\.?\d+)?)分', page5_text)
+        if total_em:
+            result['情绪总分'] = total_em.group(1)
+    
+    # ===== 大五人格 (第7页/索引6) =====
+    if page_texts and len(page_texts) > 6:
+        page7_text = page_texts[6]
+        
+        # 方法1: 直接匹配 "您在"XX"上的得分是 X.X 分"
+        big5_names = ['开放性', '宜人性', '责任心', '外倾性', '神经质']
+        for dim in big5_names:
+            b5_match = re.search(rf'您在[""「]{re.escape(dim)}[""」].*?得分是\s*(\d+\.?\d*)\s*分', page7_text)
+            if b5_match:
+                result[dim] = b5_match.group(1)
+        
+        # 方法2: 小数点回退 (旧方法)
+        if len([k for k in result if k in big5_names]) < 5:
+            decimal_pattern = re.findall(r'\b(\d+\.\d+)\b', page7_text)
+            valid_decimals = [d for d in decimal_pattern if 1.0 <= float(d) <= 10.0]
+            for i, dim_name in enumerate(big5_names):
+                if i < len(valid_decimals) and dim_name not in result:
+                    result[dim_name] = valid_decimals[i]
+    
+    # ===== 社会关系 (第9页/索引8) =====
+    if page_texts and len(page_texts) > 8:
+        page9_text = page_texts[8]
+        all_num_matches = re.findall(r'(\d+)\s*分', page9_text)
+        valid_nums = [n for n in all_num_matches if 1 <= int(n) <= 100]
+        
+        if len(valid_nums) >= 9:
+            # 页面9段落中数字按子维度分组(信赖→沟通→亲近), 每组内按母亲→父亲→同伴排列
+            # "信赖...37分,40分,34分;沟通上...32分,32分,25分;亲近上...15分,16分,17分"
+            result['与母亲信任'] = valid_nums[0]
+            result['与父亲信任'] = valid_nums[1]
+            result['与同伴信任'] = valid_nums[2]
+            result['与母亲沟通'] = valid_nums[3]
+            result['与父亲沟通'] = valid_nums[4]
+            result['与同伴沟通'] = valid_nums[5]
+            result['与母亲亲近'] = valid_nums[6]
+            result['与父亲亲近'] = valid_nums[7]
+            result['与同伴亲近'] = valid_nums[8]
+    
+    # ===== 身体健康 (第11页/索引10) =====
+    # 格式: "BMI:22kg/m²" "身高:156cm" "体重:53kg" "10小时/周" "9小时/每天"
+    if page_texts and len(page_texts) > 10:
+        page11_text = page_texts[10]
+        
+        # BMI格式: "BMI:22kg/m²"
+        bmi_match = re.search(r'BMI[::]*\s*(\d+(?:\.\d+)?)\s*kg', page11_text)
+        if bmi_match:
+            result['BMI'] = bmi_match.group(1)
+        
+        # 身高: "身高:156cm"
+        height_match = re.search(r'身高[::]*\s*(\d+)\s*cm', page11_text)
+        if height_match:
+            result['身高'] = height_match.group(1)
+        
+        # 体重: "体重:53kg"
+        weight_match = re.search(r'体重[::]*\s*(\d+(?:\.\d+)?)\s*kg', page11_text)
+        if weight_match:
+            result['体重'] = weight_match.group(1)
+        
+        # 睡眠: "10小时/周" 或 "9小时/每天" (先找周再找天)
+        sleep_week = re.search(r'(\d+)\s*小时\s*/\s*周', page11_text)
+        if sleep_week:
+            result['睡眠_小时'] = sleep_week.group(1)
+        else:
+            sleep_day = re.search(r'(\d+)\s*小时\s*/\s*每天', page11_text)
+            if sleep_day:
+                result['睡眠_小时'] = sleep_day.group(1)
+        
+        # 饮食: "9小时/每天"
+        diet = re.search(r'(\d+)\s*小时\s*/\s*每天', page11_text)
+        if diet and '饮食' not in result:
+            result['饮食_小时'] = diet.group(1)
+        
+        # 运动: 找"运动习惯"附近的"X小时/周"
+        exercise = re.search(r'运动.*?(\d+)\s*小时\s*/\s*周', page11_text, re.DOTALL)
+        if exercise:
+            result['运动_小时'] = exercise.group(1)
+    
+    return result
+
+def extract_b2_data(text, page_texts=None):
+    """提取B2报告数据 - 儿童自我与家庭教养"""
+    result = {}
+    result.update(extract_score_and_percentile(text))
+    
+    # ===== 自我概念 (6维度) =====
+    # 格式: "行为表现 9分 能力与学校表现 8分 躯体外貌 9分..."
+    # 注意: 实际PDF中维度的准确名称
+    self_concept_dims = [
+        ('行为表现', '行为表现'),
+        ('能力与学校表现', '能力与学校'),  # PDF用"能力与学校表现"
+        ('躯体外貌', '躯体外貌'),
+        ('情绪状态', '情绪状态'),
+        ('合群', '合群'),
+        ('幸福与满足', '幸福与满足'),
+    ]
+    for dim_pdf, dim_out in self_concept_dims:
+        sc_match = re.search(rf'{re.escape(dim_pdf)}\s*(\d+)\s*分', text)
+        if sc_match:
+            result[f'自我概念_{dim_out}'] = sc_match.group(1)
+        else:
+            # 后备: 包含匹配
+            sc_match2 = re.search(rf'{re.escape(dim_pdf)}.*?(\d+)\s*分', text[:2000])
+            if sc_match2:
+                result[f'自我概念_{dim_out}'] = sc_match2.group(1)
+    
+    # ===== 儿童行为 (8维度) =====
+    # 格式: "Conduct problems  1分 优秀  情绪问题 Emotional state  2分 良好..."
+    behavior_dims_cn = ['品行问题', '情绪问题', '学习问题', '社交问题', '生活习惯', '多动倾向', '刻板行为', '拖延行为']
+    for dim in behavior_dims_cn:
+        b_match = re.search(rf'{re.escape(dim)}\s*(\d+)\s*分', text)
+        if b_match:
+            result[f'行为_{dim}'] = b_match.group(1)
+    
+    # 备选: 英文关键词
+    behavior_eng = [
+        (r'Conduct problems\s+(\d+)分', '品行问题'),
+        (r'Emotional state\s+(\d+)分', '情绪问题'),
+        (r'Learning situation\s+(\d+)分', '学习问题'),
+        (r'Social situation\s+(\d+)分', '社交问题'),
+        (r'Habits.*?customs\s+(\d+)分', '生活习惯'),
+        (r'Hyperactivity\s+(\d+)分', '多动倾向'),
+        (r'Stereotypic\s+(\d+)分', '刻板行为'),
+        (r'Procrastination\s+(\d+)分', '拖延行为'),
+    ]
+    for eng_pattern, cn_name in behavior_eng:
+        if f'行为_{cn_name}' not in result:
+            eng_match = re.search(eng_pattern, text)
+            if eng_match:
+                result[f'行为_{cn_name}'] = eng_match.group(1)
+    
+    # ===== 家庭环境 (10维度) =====
+    # 格式: "亲密9分 情感表达7分 和谐5分..."
+    family_dims = ['亲密', '情感表达', '和谐', '独立性', '成就向导', 
+                   '文化氛围', '娱乐活动', '道德观念', '家务安排', '家庭规则']
+    # 方法1: 逐个匹配 "维度名+数字+分"
+    for dim in family_dims:
+        f_match = re.search(rf'{re.escape(dim)}\s*(\d+)\s*分', text)
+        if f_match:
+            result[f'家庭_{dim}'] = f_match.group(1)
+    
+    # 方法2: 在家庭环境相关页面批量提取数字序列
+    if len([k for k in result if k.startswith('家庭_')]) < 5:
+        if page_texts and len(page_texts) > 5:
+            page6_text = page_texts[5]
+            family_nums = re.findall(r'(\d+)\s*分', page6_text)
+            valid = [n for n in family_nums if 1 <= int(n) <= 15]
+            if len(valid) >= 10:
+                for i, dim in enumerate(family_dims):
+                    if f'家庭_{dim}' not in result and i < len(valid):
+                        result[f'家庭_{dim}'] = valid[i]
+    
+    return result
+
+def extract_b3_data(text, page_texts=None):
+    """提取B3报告数据 - 核心学习能力(执行功能+学习动机+学习策略)"""
+    result = {}
+    result.update(extract_score_and_percentile(text))
+    
+    # ===== 执行功能 (3维度, 百分位) =====
+    # 格式: "抑制控制 94%  工作记忆 89%  认知灵活性 60%"
+    exec_dims = ['抑制控制', '工作记忆', '认知灵活性']
+    for dim in exec_dims:
+        ef_match = re.search(rf'{re.escape(dim)}\s*(\d+)%', text)
+        if ef_match:
+            result[f'{dim}_pct'] = ef_match.group(1)
+    
+    # ===== 学习动机 (3维度, 十分制) =====
+    # 格式: "深层动机\n我的得分:8分" 或 "深层动机 8分"
+    motivation_dims = ['深层动机', '表面动机', '自我效能感']
+    for dim in motivation_dims:
+        # 方法1: "深层动机...我的得分:8分"
+        lm_match = re.search(
+            rf'{re.escape(dim)}.*?我的得分[::]\s*(\d+(?:\.\d+)?)',
+            text,
+            re.DOTALL
+        )
+        if lm_match:
+            result[dim] = lm_match.group(1)
+        else:
+            # 方法2: "深层动机 8分"
+            lm_match2 = re.search(rf'{re.escape(dim)}\s*(\d+(?:\.\d+)?)\s*分', text)
+            if lm_match2:
+                result[dim] = lm_match2.group(1)
+    
+    # ===== 学习策略 (3维度, 十分制) =====
+    # 格式: "深层方法与策略...我的得分:6.8分"
+    strategy_dims = ['深层方法与策略', '表面方法与策略', '学习自我调节']
+    for dim in strategy_dims:
+        ls_match = re.search(
+            rf'{re.escape(dim)}.*?我的得分[::]\s*(\d+(?:\.\d+)?)',
+            text,
+            re.DOTALL
+        )
+        if ls_match:
+            result[dim] = ls_match.group(1)
+        else:
+            ls_match2 = re.search(rf'{re.escape(dim)}\s*(\d+(?:\.\d+)?)\s*分', text)
+            if ls_match2:
+                result[dim] = ls_match2.group(1)
+    
+    return result
+
+# ── 成长型思维仪表盘连续分数(基于表盘指针位置)──
+# 方法:从Layout B B4报告第12页提取1090×320半圆形仪表盘图像,
+# 测量三角指针的指向角度,通过atan2映射到0-100连续分数。
+# 校准基准:陈嘉梁(成长型倾向)指针角度≈-6.68°→score=44.0
+
+
+def _extract_1090x320_image(doc):
+    """从PDF文档中提取1090×320仪表盘图像的RGB像素数据。
+    
+    Returns:
+        PIL.Image对象 (RGB模式) 或 None
+    """
+    try:
+        page = doc[12]
+        for img_info in page.get_images():
+            xref = img_info[0]
+            data = doc.extract_image(xref)
+            w, h = data['width'], data['height']
+            if (w, h) == (1090, 320):
+                img = Image.open(io.BytesIO(data['image']))
+                if img.mode != 'RGB':
+                    img = img.convert('RGB')
+                return img
+    except Exception:
+        pass
+    return None
+
+
+def _measure_needle_position(img, center_x=544):
+    """从1090×320仪表盘图像中检测三角指针位置。
+    
+    原理:指针在图像中表现为左右刻度弧线之间的第三段彩色区域。
+    从上往下扫描(y=60-220),找到指针与弧线分离最清晰的y坐标,
+    返回指针横截面中心位置。
+    
+    Returns:
+        (needle_x, needle_y, needle_width) 或 (None, None, None)
+    """
+    W, H = img.size
+    
+    for y in range(60, 220):
+        runs = []
+        in_run = False
+        run_start = 0
+        x_min = max(0, center_x - 250)
+        x_max = min(W, center_x + 250)
+        for x in range(x_min, x_max):
+            p = img.getpixel((x, y))
+            if p[0] > 30 or p[1] > 30 or p[2] > 30:
+                if not in_run:
+                    run_start = x
+                    in_run = True
+            else:
+                if in_run:
+                    runs.append((run_start, x - 1))
+                    in_run = False
+        if in_run:
+            runs.append((run_start, x_max - 1))
+        
+        big = [(s, e) for s, e in runs if e - s + 1 > 5]
+        
+        if len(big) == 3:
+            left_end = big[0][1]
+            needle = big[1]
+            right_start = big[2][0]
+            if left_end < needle[0] and needle[1] < right_start:
+                needle_mid = (needle[0] + needle[1]) / 2.0
+                needle_width = needle[1] - needle[0] + 1
+                return needle_mid, y, needle_width
+    
+    return None, None, None
+
+
+def compute_growth_mindset_score_from_needle(doc):
+    """从PDF文档的仪表盘图像计算成长型思维连续分数。
+    
+    三步流程:
+    1. 提取1090×320仪表盘图像
+    2. 检测三角指针位置(3段分离法)
+    3. 计算指针角度并映射到0-100连续分数
+    
+    校准基准:
+    - 陈嘉梁(成长型倾向44分): 指针中心偏移center_x=-18.5px, y=97
+      → atan2(-18.5, 193.8-97) = atan2(-18.5, 96.8) = -10.8°
+      → score = (-10.8 + 90) / 180 * 100 = 44.0
+    
+    Returns:
+        float 分数(0-100) 或 None(无法检测)
+    """
+    img = _extract_1090x320_image(doc)
+    if img is None:
+        return None
+    
+    # 固定参数:表盘中心x坐标和指针旋转中心y坐标(从陈嘉梁标定)
+    center_x = 544
+    center_y = 193.8  # 指针旋转中心y坐标(由陈嘉梁44分标定)
+    
+    needle_x, needle_y, _ = _measure_needle_position(img, center_x)
+    if needle_x is None:
+        return None
+    
+    dx = needle_x - center_x
+    dy = center_y - needle_y
+    if dy <= 0:
+        return None
+    
+    angle_rad = math.atan2(dx, dy)
+    angle_deg = math.degrees(angle_rad)
+    
+    # 映射到0-100: 角度-90°(左) → 0分, 0°(上) → 50分, 90°(右) → 100分
+    score = (angle_deg + 90) / 180.0 * 100.0
+    return max(0.0, min(100.0, round(score, 1)))
+
+
+def extract_growth_mindset_from_b4_vec(pdf_path):
+    """分析B4 PDF矢量图形提取成长型思维倾向(Layout B第12页的复选框+条形图颜色)
+    
+    研究发现: Layout B (16页) 的113份文件中:
+    - 59份: 第12页为纯说明性内容, 无矢量复选框 → 无成长型思维数据
+    - 43份: 复选框在LEFT(固定型)teal高亮, RIGHT(成长型)黄色 → 固定型倾向
+    - 11份: 复选框在LEFT黄色, RIGHT teal高亮 → 成长型倾向
+    
+    参数:
+        pdf_path: PDF文件路径
+    
+    返回: '固定型倾向' / '成长型倾向' / '' (无数据)
+    """
+    try:
+        doc = fitz.open(pdf_path)
+        # 仅Layout B (16页) 可能有生长型思维矢量图形
+        if len(doc) != 16:
+            doc.close()
+            return ''
+        
+        page = doc[12]  # 成长型思维页面 (Layout B)
+        paths = page.get_drawings()
+        
+        # 1) 检测teal色条形图宽度 (3种: 103/135/151)
+        bar_width = 0
+        for path in paths:
+            fill = path.get('fill')
+            rect = path.get('rect')
+            if not fill:
+                continue
+            w = rect[2] - rect[0]
+            if w > 80 and fill[0] > 0.1 and fill[0] < 0.3 and fill[1] > 0.6:
+                bar_width = round(w)
+                break
+        
+        if bar_width == 0 or bar_width == 135:
+            # 135 = 没有复选框的Layout变体 → 无个性化数据
+            doc.close()
+            return ''
+        
+        # 2) 检测复选框颜色 (48x17的矩形)
+        # LEFT侧 (x~79) = 固定型思维特征; RIGHT侧 (x~467) = 成长型思维特征
+        teal_on_left = 0
+        teal_on_right = 0
+        for path in paths:
+            fill = path.get('fill')
+            rect = path.get('rect')
+            if not fill:
+                continue
+            w = round(rect[2] - rect[0])
+            h = round(rect[3] - rect[1])
+            if w == 48 and h == 17:
+                if rect[0] < 200:  # LEFT = 固定型
+                    if fill[0] > 0.1 and fill[0] < 0.3 and fill[1] > 0.6:
+                        teal_on_left += 1
+                elif rect[0] > 300:  # RIGHT = 成长型
+                    if fill[0] > 0.1 and fill[0] < 0.3 and fill[1] > 0.6:
+                        teal_on_right += 1
+        
+        # 判断: 合并条形图宽度和复选框证据
+        result = ''
+        if teal_on_left >= 3 and teal_on_right == 0:
+            result = '固定型倾向'
+        elif teal_on_right >= 3 and teal_on_left == 0:
+            result = '成长型倾向'
+        elif bar_width == 151:
+            result = '固定型倾向'
+        elif bar_width == 103:
+            result = '成长型倾向'
+        
+        doc.close()
+        return result
+    except Exception as e:
+        print(f"Error extracting growth mindset from {pdf_path}: {e}")
+        return ''
+
+
+def extract_b4_data(text, page_texts=None, pdf_path=None):
+    """提取B4报告数据 - 核心认知能力+自我概念+自驱力+成长型思维"""
+    result = {}
+    result.update(extract_score_and_percentile(text))
+    
+    # ===== 核心认知维度 (同A1, 6维度) =====
+    cognitive_dims = ['感知觉', '注意力', '记忆力', '推理能力', '空间能力', '加工速度']
+    
+    # 格式2 (优先): summary认知模型页面 "感知觉 |  Perception\n...描述...\n百分位(%)\n14"
+    # 该页面百分位数据最准确, 不会被detail页面的刻度标签(0%-100%)干扰
+    for dim in cognitive_dims:
+        pct_match = re.search(
+            rf'{re.escape(dim)}\s*\|\s*\w+[\s\S]*?百分位(%)\s*\n\s*(\d+)',
+            text
+        )
+        if pct_match:
+            result[f'{dim}_pct'] = pct_match.group(1)
+    
+    # 格式1 (后补): detail页面, 仅当格式2未提取到时使用
+    #  格式1a: "我的感知觉得分\n17|18%" 或 带乱码分隔符 "17ح18%"
+    #  格式1b: "我的加工速度得分\n33 | 33 | 88%" (原始分|标准分|百分位)
+    #  使用两步法避免回溯bug: 先截取"得分"后文本段, 再分别搜score和pct
+    #  注意: 11页B4只有推理能力/空间能力/加工速度3维有detail页, 所以只补这3维
+    for dim in cognitive_dims:
+        detail_section = re.search(
+            rf'我的{re.escape(dim)}得分(.{{1,300}})',
+            text, re.DOTALL
+        )
+        if detail_section:
+            section_text = detail_section.group(1)
+            # 提取第一个数字%作为百分位 (首个出现的是实际分数)
+            pct_match = re.search(r'(\d+)%', section_text)
+            if pct_match:
+                if f'{dim}_pct' not in result:
+                    result[f'{dim}_pct'] = pct_match.group(1)
+            # 提取第一个数字作为原始分
+            score_match = re.search(r'(\d+)', section_text)
+            if score_match:
+                result[f'{dim}_score'] = score_match.group(1)
+    
+    # ===== 自我概念 (6维度, 0-10分制, 第11页柱状图) =====
+    # B4第11页(索引10)是横向柱状图, 每维度一个横条, 0-10刻度
+    # 分数印制在横条右端上方, font_size≈9.6(区别于坐标轴标签6.2和维度名称8.6)
+    # 使用positioned text提取, 禁止使用%匹配(会拿到认知页的百分位值)
+    self_concept_out = ['行为表现', '能力与学校', '躯体外貌', '情绪状态', '合群', '幸福与满足']
+    if pdf_path:
+        try:
+            doc = fitz.open(pdf_path)
+            n_pages = len(doc)
+            # 两种B4格式:
+            #   16页格式: self-concept在第11页(索引10), 分数font_size≈9.6
+            #   11页格式: self-concept在第6页(索引5), 分数font_size≈6.4
+            candidates = [(10, 8, 11), (5, 6, 8)] if n_pages > 12 else [(5, 6, 8), (10, 8, 11)]
+            found = False
+            for page_idx, fs_min, fs_max in candidates:
+                if page_idx >= n_pages:
+                    continue
+                page = doc[page_idx]
+                blocks = page.get_text('dict')['blocks']
+                scores = []
+                for b in blocks:
+                    if b['type'] == 0:
+                        for line in b['lines']:
+                            for span in line['spans']:
+                                span_text = span['text'].strip()
+                                if not span_text:
+                                    continue
+                                size = span['size']
+                                x = span['bbox'][0]
+                                if (fs_min <= size <= fs_max and x > 250
+                                    and span_text.isdigit() and 0 <= int(span_text) <= 10):
+                                    scores.append((span['bbox'][1], int(span_text)))
+                scores.sort(key=lambda s: s[0])
+                if len(scores) >= 6:
+                    for i, dim in enumerate(self_concept_out):
+                        result[f'自我概念_{dim}'] = str(scores[i][1])
+                    found = True
+                    break
+            if not found:
+                # 第三次尝试: 在所有页面中搜索"自我概念"字样, 提取相邻数字
+                for pi in range(n_pages):
+                    page_text = doc[pi].get_text()
+                    if '自我概念' in page_text and '行为表现' in page_text:
+                        blocks = doc[pi].get_text('dict')['blocks']
+                        scores = []
+                        for b in blocks:
+                            if b['type'] == 0:
+                                for line in b['lines']:
+                                    for span in line['spans']:
+                                        st = span['text'].strip()
+                                        if not st:
+                                            continue
+                                        size = span['size']
+                                        x = span['bbox'][0]
+                                        if (5 <= size <= 12 and x > 250
+                                            and st.isdigit() and 0 <= int(st) <= 10):
+                                            scores.append((span['bbox'][1], int(st)))
+                        scores.sort(key=lambda s: s[0])
+                        if len(scores) >= 6:
+                            for i, dim in enumerate(self_concept_out):
+                                result[f'自我概念_{dim}'] = str(scores[i][1])
+                        break
+            doc.close()
+        except Exception as e:
+            print(f"  [WARN] B4 self-concept extraction failed: {e}")
+    
+    # ===== 自驱力 =====
+    all_scores = re.findall(r'我的得分[::](\d+\.?\d*)', text)
+    if len(all_scores) >= 3:
+        result['自主性'] = all_scores[0]
+        result['胜任感'] = all_scores[1]
+        result['归属感'] = all_scores[2]
+    elif len(all_scores) >= 1:
+        result['自主性'] = all_scores[0]
+    
+    if '自主性' not in result:
+        parts = text.split('Autonomy')
+        if len(parts) >= 3:
+            match = re.search(r'(\d+\.?\d*)', parts[2])
+            if match:
+                result['自主性'] = match.group(1)
+    
+    if '胜任感' not in result:
+        parts = text.split('Competence')
+        if len(parts) >= 3:
+            match = re.search(r'(\d+\.?\d*)', parts[3] if len(parts) > 3 else parts[2])
+            if match:
+                result['胜任感'] = match.group(1)
+    
+    if '归属感' not in result:
+        parts = text.split('Relatedness')
+        if len(parts) >= 3:
+            match = re.search(r'(\d+\.?\d*)', parts[3] if len(parts) > 3 else parts[2])
+            if match:
+                result['归属感'] = match.group(1)
+    
+    # ===== 成长型思维(连续分数) =====
+    # 从1090×320仪表盘图像检测指针位置,映射到0-100刻度
+    # Layout A (11页) 的B4文件无仪表盘图像 → 留空
+    result['成长型思维'] = ''
+    
+    if pdf_path:
+        try:
+            doc = fitz.open(pdf_path)
+            score = compute_growth_mindset_score_from_needle(doc)
+            doc.close()
+            if score is not None:
+                result['成长型思维'] = score
+        except Exception:
+            pass
+    
+    return result
+
+def extract_b6_data(text, page_texts=None):
+    """提取B6报告数据 - 职业发展(兴趣+能力+价值观)"""
+    result = {}
+    result.update(extract_score_and_percentile(text))
+    
+    # ===== 职业兴趣 (Holland 6类型) =====
+    # 格式: "NO.1: 艺术型 Artistic NO.2: 社会型 Social 7分 5分"
+    interest_dims = {
+        '艺术型': '兴趣_艺术型',
+        '社会型': '兴趣_社会型',
+        '事业型': '兴趣_事业型',
+        '常规型': '兴趣_常规型',
+        '现实型': '兴趣_现实型',
+        '研究型': '兴趣_研究型',
+    }
+    
+    # 方法1: NO.X模式
+    for cn_name, field_name in interest_dims.items():
+        # Match "X型 ... X分" where X分 follows the interest type name
+        h_match = re.search(
+            rf'{re.escape(cn_name)}\s+\w+\s+(\d+)\s*分',
+            text
+        )
+        if h_match:
+            result[field_name] = h_match.group(1)
+    
+    # ===== 多元智能 (8能力) =====
+    # 格式: "内省能力 INTRAPERSONAL 8分  人际关系能力 INTERPERSONAL 4分"
+    ability_dims = {
+        '内省能力': '能力_内省',
+        '空间能力': '能力_空间',
+        '音乐能力': '能力_音乐',
+        '人际关系能力': '能力_人际关系',
+        '自然能力': '能力_自然',
+        '身体运动能力': '能力_身体运动',
+        '语言能力': '能力_语言',
+        '逻辑数学能力': '能力_逻辑数学',
+    }
+    
+    for cn_name, field_name in ability_dims.items():
+        ab_match = re.search(
+            rf'{re.escape(cn_name)}\s+\w+\s+(\d+)\s*分',
+            text
+        )
+        if ab_match:
+            result[field_name] = ab_match.group(1)
+        else:
+            # 更简单: "内省能力 8分"
+            ab_match2 = re.search(
+                rf'{re.escape(cn_name)}\s*(\d+)\s*分',
+                text
+            )
+            if ab_match2:
+                # Make sure it's not matching a multi-digit number in a different context
+                score = ab_match2.group(1)
+                if 1 <= int(score) <= 15:
+                    result[field_name] = score
+    
+    # ===== 职业价值观 (可选) =====
+    # 格式: 在"我的职业价值观"区域有因子得分如 "8.11\n声望地位" "5.41\n美的追求"
+    # 取第一个因子得分作为代表性分数(限1-15分)
+    if '我的职业价值观' in text:
+        m = re.search(r'我的职业价值观.*?(\d+(?:\.\d+)?)', text, re.DOTALL)
+        if m:
+            val = float(m.group(1))
+            if 1 <= val <= 15:
+                result['职业价值观'] = m.group(1)
+    # 后备: "最高分"附近的明确数值
+    if '职业价值观' not in result:
+        m = re.search(r'职业价值观.*?最高分[^\d]*?(\d+(?:\.\d+)?)', text, re.DOTALL)
+        if m:
+            val = float(m.group(1))
+            if 1 <= val <= 15:
+                result['职业价值观'] = m.group(1)
+    
+    return result
+
+def extract_c1_data(text, page_texts=None):
+    """提取C1报告数据 - 校园版综合(认知+人格+自驱力+自我概念)"""
+    result = {}
+    result.update(extract_score_and_percentile(text))
+    
+    # ===== 核心认知维度 (同A1, 6维度) =====
+    cognitive_dims = ['感知觉', '注意力', '记忆力', '推理能力', '空间能力', '加工速度']
+    
+    # 格式1: "我的感知觉得分 41  | 14%"
+    for dim in cognitive_dims:
+        detail_match = re.search(
+            rf'我的{re.escape(dim)}得分\s*(\d+)\s*\|\s*(\d+)%',
+            text
+        )
+        if detail_match:
+            result[f'{dim}_score'] = detail_match.group(1)
+            result[f'{dim}_pct'] = detail_match.group(2)
+    
+    # 格式2: summary页面 "感知觉 | Perception\n百分位(%)\n14"
+    for dim in cognitive_dims:
+        if f'{dim}_pct' not in result:
+            pct_match = re.search(
+                rf'{re.escape(dim)}\s*\|\s*\w+\s*\n\s*百分位(%)\s*\n\s*(\d+)',
+                text
+            )
+            if pct_match:
+                result[f'{dim}_pct'] = pct_match.group(1)
+    
+    # ===== 大五人格 (同A2) =====
+    big5_names = ['开放性', '宜人性', '责任心', '外倾性', '神经质']
+    if page_texts:
+        for pt in page_texts:
+            for dim in big5_names:
+                if dim not in result:
+                    b5_match = re.search(rf'您在[""「]{re.escape(dim)}[""」].*?得分是\s*(\d+(?:\.\d+)?)\s*分', pt)
+                    if b5_match:
+                        result[dim] = b5_match.group(1)
+    
+    # ===== 自驱力 (同B4) =====
+    all_scores = re.findall(r'我的得分[::](\d+\.?\d*)', text)
+    if len(all_scores) >= 3:
+        result['自主性'] = all_scores[0]
+        result['胜任感'] = all_scores[1]
+        result['归属感'] = all_scores[2]
+    elif len(all_scores) >= 1:
+        result['自主性'] = all_scores[0]
+    
+    # ===== 自我概念 (同B4) =====
+    self_concept_start = text.find('SELF-CONCEPT')
+    if self_concept_start >= 0:
+        before_section = text[:self_concept_start]
+        pct_matches = re.findall(r'(\d+)%', before_section)
+        
+        sc_dims = ['行为表现', '能力与学校', '躯体外貌', '情绪状态', '合群', '幸福与满足']
+        if len(pct_matches) >= 12:
+            high_pcts = pct_matches[-12:]
+            for i, dim in enumerate(sc_dims):
+                idx = i * 2 + 1
+                if idx < len(high_pcts):
+                    result[f'自我概念_{dim}'] = high_pcts[idx]
+        elif len(pct_matches) >= 6:
+            for i, dim in enumerate(sc_dims):
+                if i < len(pct_matches):
+                    result[f'自我概念_{dim}'] = pct_matches[-(6-i)]
+    
+    return result
+
+# ===== B5 人际关系指南针图子维度提取 =====
+# 校准: 每个象限的 fill_avg → score 线性映射系数
+# v2: 扫描起点r=150跳过中心色块; 因子由extract_b5_data传入per-student值
+_B5_QUAD_FACTORS = {
+    '母亲关系': 0.006974,
+    '父亲关系': 0.005834,
+    '师生关系': 0.006820,
+    '同伴关系': 0.005049,
+}
+_B5_SUB_DIMS = {
+    '母亲关系': ['控制压迫', '情感疏离', '价值冲突', '双重标准', '过度期待'],
+    '父亲关系': ['控制压迫', '情感疏离', '价值冲突', '双重标准', '过度期待'],
+    '师生关系': ['回避型', '敌对型', '高控制型', '关系疏离型'],
+    '同伴关系': ['回避型', '被排斥型', '攻击型', '边缘型', '特殊因素型'],
+}
+# 象限角度范围 (图像坐标系: 0°=右, 90°=下, 180°=左, 270°=上)
+_B5_QUAD_ANGLES = {
+    '母亲关系': (185, 265),
+    '父亲关系': (140, 178),
+    '师生关系': (275, 350),
+    '同伴关系': (2, 42),
+}
+_B5_CENTER = (857, 758)
+
+def _measure_fill_at_angle(img, cx, cy, angle_deg):
+    """沿指定角度测量填充范围(像素半径)。
+    从r=150开始扫描以跳过中心色块(避免环线阻断),
+    25px间隙容忍用于跨越填充分区内的环线。"""
+    angle = math.radians(angle_deg)
+    w, h = img.size
+    max_r = int(min(w-cx, cx, cy, h-cy)) - 2
+    fill_end = None
+    gap = 0
+    for r in range(150, max_r, 1):
+        x = int(cx + r * math.cos(angle))
+        y = int(cy + r * math.sin(angle))
+        if not (0 <= x < w and 0 <= y < h):
+            break
+        px = img.getpixel((x, y))
+        is_white = all(c > 230 for c in px[:3])
+        is_gray = all(155 < c < 195 for c in px[:3])
+        is_black = all(c < 20 for c in px[:3])
+        is_colored = not (is_white or is_gray or is_black)
+        if is_colored:
+            fill_end = r
+            gap = 0
+        elif fill_end is not None:
+            gap += 1
+            if gap > 25:
+                break
+    return fill_end
+
+def _extract_b5_compass_scores(doc, quad_factors=None):
+    """从B5 PDF文档提取人际关系指南针19个子维度分数。
+    返回 { '人际_母亲关系_控制压迫': score, ... }"""
+    result = {}
+    page = doc[8]  # 第9页
+    imgs = page.get_images()
+    if len(imgs) < 4:
+        return result
+    base = doc.extract_image(imgs[3][0])
+    img = Image.open(io.BytesIO(base['image']))
+    cx, cy = _B5_CENTER
+    
+    # 从页面文本提取per-student factor(如果未提供)
+    if quad_factors is None:
+        text = page.get_text()
+        lines = text.split('\n')
+        text_scores = {}
+        for rel in ['母亲关系', '父亲关系', '师生关系', '同伴关系']:
+            for i, line in enumerate(lines):
+                if line.strip() == rel:
+                    for j in range(i+1, min(i+5, len(lines))):
+                        try:
+                            text_scores[rel] = float(lines[j].strip())
+                            break
+                        except ValueError:
+                            pass
+                    break
+        # 计算quad平均fill和per-student factor
+        quad_factors = {}
+        for qname, (a1, a2) in _B5_QUAD_ANGLES.items():
+            fills = []
+            for a in range(a1, a2+1):
+                fe = _measure_fill_at_angle(img, cx, cy, a)
+                if fe:
+                    fills.append(fe)
+            avg = sum(fills)/len(fills) if fills else 1
+            txt = text_scores.get(qname, 3.0)
+            quad_factors[qname] = txt / avg if avg > 0 else _B5_QUAD_FACTORS[qname]
+    
+    for quad_name, (a_start, a_end) in _B5_QUAD_ANGLES.items():
+        dims = _B5_SUB_DIMS[quad_name]
+        n = len(dims)
+        step = (a_end - a_start) / (n + 1)
+        factor = quad_factors.get(quad_name, _B5_QUAD_FACTORS[quad_name])
+        
+        for i, dim_name in enumerate(dims):
+            angle = a_start + (i + 1) * step
+            fills = []
+            for offset in [-2, 0, 2]:
+                fe = _measure_fill_at_angle(img, cx, cy, angle + offset)
+                if fe is not None:
+                    fills.append(fe)
+            avg_fill = sum(fills) / len(fills) if fills else 0
+            score = min(5.0, max(1.0, round(avg_fill * factor, 1)))
+            result[f'人际_{quad_name}_{dim_name}'] = score
+    return result
+
+def extract_b5_data(text, page_texts=None, pdf_path=None):
+    """提取B5报告数据 - 青春期挑战(情绪调节+学业压力+人际关系+社交+睡眠+运动+网络依赖)"""
+    result = {}
+    result.update(extract_score_and_percentile(text))
+    
+    # ===== 情绪调节策略 (2维度) =====
+    er_match = re.search(r'认知重评.*?Cognitive Reappraisal\s*(\d+(?:\.\d+)?)\s*分', text, re.DOTALL)
+    if not er_match:
+        er_match = re.search(r'认知重评\s*(\d+(?:\.\d+)?)\s*分', text)
+    if er_match:
+        result['认知重评'] = er_match.group(1)
+    
+    es_match = re.search(r'表达抑制.*?Expressive Suppression\s*(\d+(?:\.\d+)?)\s*分', text, re.DOTALL)
+    if not es_match:
+        es_match = re.search(r'表达抑制\s*(\d+(?:\.\d+)?)\s*分', text)
+    if es_match:
+        result['表达抑制'] = es_match.group(1)
+    
+    # ===== 学业压力源 (5维度) =====
+    stress_section = re.search(r'学业压力源[\s\S]*?(?=人际关系问题)', text)
+    if not stress_section:
+        stress_section = re.search(r'学业压力[\s\S]*?(?=社交能力|睡眠)', text)
+    stress_text = stress_section.group() if stress_section else text
+    stress_dims = ['学业负担', '家庭期望', '师生关系', '自我期望', '同伴竞争']
+    for dim in stress_dims:
+        m = re.search(re.escape(dim) + r'\s*\n\s*(\d+(?:\.\d+)?)', stress_text)
+        if m:
+            result[f'学业压力_{dim}'] = m.group(1)
+    
+    # ===== 人际关系问题 (4维度) =====
+    rel_section = re.search(r'人际关系问题[\s\S]*?(?=社交能力)', text)
+    rel_text = rel_section.group() if rel_section else text
+    relation_dims = ['母亲关系', '父亲关系', '师生关系', '同伴关系']
+    # 数据区格式: "维度名\n描述句。\n分数" 需跳过intro段中的同名词
+    for dim in relation_dims:
+        m = re.search(re.escape(dim) + r'\n[^\n]*。\n(\d+(?:\.\d+)?)', rel_text)
+        if m:
+            result[f'人际_{dim}'] = m.group(1)
+    
+    # ===== 社交能力 (5维度) =====
+    soc_section = re.search(r'社交能力[\s\S]*?(?=睡眠)', text)
+    soc_text = soc_section.group() if soc_section else text
+    soc_m = re.search(r'同龄人平均水平\s*\n\s*(\d+(?:\.\d+)?)\s*\n\s*(\d+(?:\.\d+)?)\s*\n\s*(\d+(?:\.\d+)?)\s*\n\s*(\d+(?:\.\d+)?)\s*\n\s*(\d+(?:\.\d+)?)', soc_text)
+    if soc_m:
+        soc_dims = ['主动交往', '情感支持', '情感表达', '表达影响', '冲突解决']
+        for i, dim in enumerate(soc_dims):
+            result[f'社交_{dim}'] = soc_m.group(i+1)
+    
+    # ===== 睡眠数据 =====
+    sleep_hours = re.search(r'睡眠时长[^。]*?(\d+(?:\.\d+)?)\s*小时', text)
+    if sleep_hours:
+        result['睡眠_时长'] = sleep_hours.group(1)
+    
+    sleep_eff = re.search(r'睡眠效率[^。]*?(\d+(?:\.\d+)?)%', text)
+    if sleep_eff:
+        result['睡眠_效率'] = sleep_eff.group(1)
+    
+    sleep_lat = re.search(r'入睡时间[^。]*?(\d+)-(\d+)分钟', text)
+    if sleep_lat:
+        result['入睡时间_min'] = sleep_lat.group(1)
+    
+    # 睡眠定性维度:取维度名后第三行(值行)
+    sleep_qual = {
+        '主观睡眠质量': '睡眠_主观质量',
+        '日间功能障碍': '睡眠_日间功能',
+        '催眠药物': '睡眠_催眠药物',
+    }
+    for kw, col in sleep_qual.items():
+        m = re.search(re.escape(kw) + r'\n[^\n]*\n[^\n]*\n([^\n]+)', text)
+        if m:
+            result[col] = m.group(1).strip()
+    
+    sleep_disorder = re.search(r'睡眠障碍[^。]*?(\S+)', text)
+    if sleep_disorder:
+        result['睡眠_障碍'] = sleep_disorder.group(1).strip()
+    
+    # ===== 运动数据 =====
+    ex_names = ['久坐', '步行', '中等强度', '高强度']
+    for ex_name in ex_names:
+        m = re.search(re.escape(ex_name) + r'\n(\d+)\n天/周\n(.+)', text)
+        if m:
+            result[f'运动_{ex_name}_频率'] = m.group(1)
+            result[f'运动_{ex_name}_时长'] = m.group(2).strip()
+    
+    # ===== 网络依赖 =====
+    internet_total = re.search(r'(\d+)分\s*\n\s*强迫使用', text)
+    if internet_total:
+        result['网络依赖_总分'] = internet_total.group(1)
+    
+    # 用上方的分数匹配子维度(分数在维度名之上)
+    # 格式: "44分\n强迫使用\n...43分\n时间管理问题\n...35分\n戒断反应"
+    internet_sub = re.findall(r'(\d+)分\s*\n\s*(强迫使用|时间管理问题|戒断反应)', text)
+    for score, dim_name in internet_sub:
+        col_map = {'强迫使用': '网络依赖_强迫使用', '时间管理问题': '网络依赖_时间管理', '戒断反应': '网络依赖_戒断反应'}
+        if dim_name in col_map:
+            result[col_map[dim_name]] = score
+    
+    # ===== 人际关系指南针19个子维度(第9页图像) =====
+    if pdf_path and os.path.exists(pdf_path):
+        try:
+            doc = fitz.open(pdf_path)
+            if len(doc) > 8:
+                compass_scores = _extract_b5_compass_scores(doc)
+                result.update(compass_scores)
+            doc.close()
+        except Exception:
+            pass  # 图像提取失败不影响已有数据
+    
+    return result
+
+def determine_report_type(filename):
+    """根据文件名判断报告类型(后备方案,内容指纹优先)"""
+    # 已被 type_detector.detect_type() 替代
+    # 保留作为纯后备(当文本层损坏导致 detect_type 回退时调用 fallback_by_filename)
+    from type_detector import fallback_by_filename
+    return fallback_by_filename(filename)
+
+def extract_date_from_filename(filename):
+    """从文件名中提取测评日期 YYYYMMDD"""
+    # 匹配 8位连续数字日期 (如 20250320)
+    m = re.search(r'(\d{8})', filename)
+    if m:
+        return m.group(1)
+    # 匹配 YYYY-MM-DD 或 YYYY_MM_DD 格式(文件名中常见带连字符)
+    m = re.search(r'(\d{4})[-_](\d{2})[-_](\d{2})', filename)
+    if m:
+        return m.group(1) + m.group(2) + m.group(3)
+    return ''
+
+def extract_all_data(pdf_path, filename):
+    """提取所有类型报告的数据"""
+    text = extract_text_from_pdf(pdf_path)
+    page_texts = extract_text_from_pdf(pdf_path, return_pages=True)
+    
+    name = extract_name_from_filename(filename)
+    birthday = extract_birthday_from_pdf(text, page_texts)
+    report_type, _ = detect_type(text, filename)
+    
+    # 基础信息
+    row = {
+        'filename': filename,
+        '姓名': name,
+        '生日': birthday,
+        '报告类型': report_type,
+        '测评日期': extract_date_from_filename(filename),
+    }
+    
+    # 根据类型提取数据
+    if report_type == 'A1':
+        data = extract_a1_data(text, page_texts)
+    elif report_type == 'A2':
+        data = extract_a2_data(text, page_texts)
+    elif report_type == 'B2':
+        data = extract_b2_data(text, page_texts)
+    elif report_type == 'B3':
+        data = extract_b3_data(text, page_texts)
+    elif report_type == 'B4':
+        data = extract_b4_data(text, page_texts, pdf_path)
+    elif report_type == 'B5':
+        data = extract_b5_data(text, page_texts, pdf_path)
+    elif report_type == 'B6':
+        data = extract_b6_data(text, page_texts)
+    elif report_type == 'C1':
+        data = extract_c1_data(text, page_texts)
+    else:
+        data = {}
+    
+    row.update(data)
+    return row
+
+def main():
+    # 自动检测项目根目录 (脚本在 scripts/ 下)
+    script_dir = os.path.dirname(os.path.abspath(__file__))
+    project_dir = os.path.dirname(os.path.dirname(script_dir))  # 项目根目录(上溯两级:scripts/reports/ → scripts/ → 根)
+    reports_dir = os.path.join(project_dir, "报告")
+    output_file = os.path.join(project_dir, "素材库", "测评数据_优化提取.csv")
+    
+    # 获取所有PDF文件
+    pdf_files = []
+    for f in os.listdir(reports_dir):
+        if f.endswith('.pdf') and '案例' not in f and '示例' not in f and 'eStatement' not in f:
+            pdf_files.append((os.path.join(reports_dir, f), f))
+    
+    print(f"Found {len(pdf_files)} PDF files")
+    
+    all_results = []
+    type_counts = {}
+    
+    for i, (pdf_path, filename) in enumerate(pdf_files):
+        if i % 100 == 0:
+            print(f"Processing {i}/{len(pdf_files)}...")
+        
+        try:
+            row = extract_all_data(pdf_path, filename)
+            rt = row.get('报告类型', 'Unknown')
+            type_counts[rt] = type_counts.get(rt, 0) + 1
+            all_results.append(row)
+        except Exception as e:
+            print(f"Error processing {filename}: {e}")
+    
+    print(f"\n=== Report Type Counts ===")
+    for rt, count in sorted(type_counts.items()):
+        print(f"{rt}: {count}")
+    
+    # 写入CSV
+    if all_results:
+        all_keys = set()
+        for row in all_results:
+            all_keys.update(row.keys())
+        
+        columns = ['filename', '姓名', '生日', '报告类型', '测评日期', '总分', '百分位']
+        for k in sorted(all_keys):
+            if k not in columns:
+                columns.append(k)
+        
+        with open(output_file, 'w', encoding='utf-8-sig', newline='') as f:
+            writer = csv.DictWriter(f, fieldnames=columns)
+            writer.writeheader()
+            writer.writerows(all_results)
+        
+        print(f"\nWritten {len(all_results)} records to {output_file}")
+
+if __name__ == "__main__":
+    main()

+ 153 - 0
docs/参考资料/dan_reports/extract_c1_names.py

@@ -0,0 +1,153 @@
+#!/usr/bin/env python
+# -*- coding: utf-8 -*-
+"""
+从 PDF 中深度提取姓名,专门处理 C1 校园版文件
+尝试方法:
+1. 从 PDF 文本内容提取(第一页前 20 行)
+2. 从 PDF 元数据提取
+3. 从 PDF 表格/结构化文本提取
+"""
+
+import os
+import re
+import pdfplumber
+from pathlib import Path
+
+def extract_name_from_pdf_deep(pdf_path):
+    """深度提取 PDF 中的姓名"""
+    try:
+        with pdfplumber.open(pdf_path) as pdf:
+            if not pdf.pages:
+                return None
+            
+            # 方法 1: 从第一页文本提取
+            text = pdf.pages[0].extract_text() or ""
+            lines = text.split('\n')
+            
+            # 在前 20 行查找姓名
+            for i, line in enumerate(lines[:20]):
+                line = line.strip()
+                if not line or len(line) < 2 or len(line) > 15:
+                    continue
+                
+                # 模式 1: 纯中文姓名 (2-4 个汉字)
+                if re.match(r'^[\u4e00-\u9fa5]{2,4}$', line):
+                    # 排除常见非姓名词汇
+                    exclude_words = ['报告', '测评', '学生', '姓名', '日期', '学校', '年级', '班级']
+                    if line not in exclude_words:
+                        return line
+                
+                # 模式 2: "姓名:XXX" 或 "姓名:XXX"
+                match = re.search(r'姓名 [::]\s*([\u4e00-\u9fa5]{2,4})', line)
+                if match:
+                    return match.group(1)
+                
+                # 模式 3: "学生:XXX"
+                match = re.search(r'学生 [::]\s*([\u4e00-\u9fa5]{2,4})', line)
+                if match:
+                    return match.group(1)
+                
+                # 模式 4: 英文名 (3-8 个字母)
+                match = re.search(r'\b([A-Z][a-z]{2,8})\b', line)
+                if match:
+                    name = match.group(1)
+                    exclude_names = ['Total', 'Score', 'Date', 'ID', 'Page', 'Report']
+                    if name not in exclude_names and name not in line.upper():
+                        return name
+            
+            # 方法 2: 尝试从 PDF 元数据提取
+            try:
+                metadata = pdf.metadata
+                if metadata:
+                    # 检查 Author, Creator, Subject 等字段
+                    for key in ['Author', 'Subject', 'Title', 'Creator']:
+                        if key in metadata and metadata[key]:
+                            val = str(metadata[key])
+                            # 尝试匹配中文姓名
+                            match = re.search(r'([\u4e00-\u9fa5]{2,4})', val)
+                            if match:
+                                return match.group(1)
+            except:
+                pass
+            
+            # 方法 3: 尝试解析表格或结构化数据
+            for page in pdf.pages[:2]:
+                tables = page.extract_tables()
+                for table in tables:
+                    for row in table[:5]:  # 前 5 行
+                        if row:
+                            for cell in row:
+                                if cell:
+                                    cell = str(cell).strip()
+                                    if re.match(r'^[\u4e00-\u9fa5]{2,4}$', cell):
+                                        exclude_words = ['报告', '测评', '学生', '姓名']
+                                        if cell not in exclude_words:
+                                            return cell
+            
+            return None
+            
+    except Exception as e:
+        return None
+
+def main():
+    # 自动计算项目根目录 (scripts/reports/ → 项目根)
+    report_dir = Path(__file__).parent.parent.parent / "报告"
+    
+    # 获取所有 C1_时间戳.pdf 格式的文件
+    c1_files = []
+    for f in report_dir.glob("C1_*.pdf"):
+        # 匹配 C1_纯数字.pdf (缺少姓名的)
+        if re.match(r'^C1_\d{16,}\.pdf$', f.name):
+            c1_files.append(f)
+    
+    print(f"找到 {len(c1_files)} 个缺少姓名的 C1 文件")
+    
+    renamed = 0
+    failed = 0
+    not_found = 0
+    
+    for file_path in c1_files:
+        filename = file_path.name
+        # 提取时间戳
+        timestamp_match = re.search(r'(\d{16,})', filename)
+        timestamp = timestamp_match.group(1) if timestamp_match else None
+        
+        if not timestamp:
+            print(f"[WARN] 无法提取时间戳:{filename}")
+            failed += 1
+            continue
+        
+        # 尝试从 PDF 提取姓名
+        name = extract_name_from_pdf_deep(str(file_path))
+        
+        if name:
+            new_filename = f"C1_{name}_{timestamp}.pdf"
+            new_path = file_path.parent / new_filename
+            
+            # 检查是否已存在
+            if new_path.exists() and new_path.resolve() != file_path.resolve():
+                base, ext = os.path.splitext(new_filename)
+                counter = 1
+                while new_path.exists():
+                    new_filename = f"{base}_{counter}{ext}"
+                    new_path = file_path.parent / new_filename
+                    counter += 1
+            
+            try:
+                file_path.rename(new_path)
+                print(f"[OK] {filename} -> {new_filename}")
+                renamed += 1
+            except Exception as e:
+                print(f"[FAIL] {filename}: {e}")
+                failed += 1
+        else:
+            print(f"[SKIP] 无法提取姓名:{filename}")
+            not_found += 1
+    
+    print(f"\n完成:")
+    print(f"  重命名:{renamed}")
+    print(f"  未找到姓名:{not_found}")
+    print(f"  失败:{failed}")
+
+if __name__ == "__main__":
+    main()

+ 308 - 0
docs/参考资料/dan_reports/merge_all_types.py

@@ -0,0 +1,308 @@
+# -*- coding: utf-8 -*-
+"""
+学生报告合并脚本 - 按姓名+1月内窗口分组
+- 同一学生30天内测评的记录合并为一次测评
+- 同类型取最新,删除过期文件
+- 不同类型合并成一条记录(丰富学生画像)
+"""
+import csv
+from collections import defaultdict
+import re
+from datetime import datetime, timedelta
+import os
+import subprocess
+
+script_dir = os.path.dirname(os.path.abspath(__file__))
+project_dir = os.path.dirname(os.path.dirname(script_dir))  # 上溯两级到项目根
+csv_path = os.path.join(project_dir, "素材库", "测评数据_优化提取.csv")
+output_path = os.path.join(project_dir, "素材库", "学生汇总数据.csv")
+
+# 读取CSV
+with open(csv_path, 'r', encoding='utf-8') as f:
+    reader = csv.DictReader(f)
+    data = list(reader)
+
+print(f'原始记录数: {len(data)}')
+
+# 过滤掉Unknown类型
+data = [r for r in data if r.get('报告类型') != 'Unknown']
+print(f'过滤Unknown后记录数: {len(data)}')
+
+def date_to_ord(d):
+    """YYYYMMDD → 儒略日数,非法返回 0"""
+    if not d or len(d) != 8:
+        return 0
+    try:
+        return datetime.strptime(d, '%Y%m%d').toordinal()
+    except:
+        return 0
+
+def normalize_date(d):
+    """统一日期格式为YYYY-MM-DD"""
+    if not d:
+        return ''
+    d = d.strip().replace('-', '').replace('/', '')
+    if len(d) == 8:
+        return f'{d[:4]}-{d[4:6]}-{d[6:8]}'
+    return d
+
+def extract_ts(filename):
+    """从文件名提取最大数字时间戳"""
+    digits = re.findall(r'\d+', filename)
+    return max(int(d) for d in digits) if digits else 0
+
+# ===== 第1步:按姓名分组 =====
+name_groups = defaultdict(list)
+for r in data:
+    name = r.get('姓名', '').strip()
+    if name:
+        r['_date'] = r.get('测评日期', '').strip()
+        name_groups[name].append(r)
+
+print(f'学生人数: {len(name_groups)}')
+
+# ===== 第2步:30天窗口聚类 =====
+clusters = []  # 每个元素 (cluster_date, 学生名, 记录列表)
+files_to_delete = []
+
+for name, records in name_groups.items():
+    # 按日期排序
+    sorted_recs = sorted(records, key=lambda x: x['_date'])
+    
+    # 窗口聚类:从第一个记录开始,30天内归入同组
+    current_cluster = []
+    for r in sorted_recs:
+        if not current_cluster:
+            current_cluster.append(r)
+        else:
+            first_date = current_cluster[0]['_date']
+            r_date = r['_date']
+            if first_date and r_date and (date_to_ord(r_date) - date_to_ord(first_date)) <= 30:
+                current_cluster.append(r)
+            else:
+                clusters.append((current_cluster[0]['_date'], name, current_cluster))
+                current_cluster = [r]
+    if current_cluster:
+        clusters.append((current_cluster[0]['_date'], name, current_cluster))
+
+print(f'聚类后组数: {len(clusters)}')
+
+# ===== 第3步:处理每个聚类 =====
+merged_students = []
+
+for cluster_date, name, recs_in_cluster in clusters:
+    # 按时间戳排序(倒序,最新在前)
+    recs_sorted = sorted(recs_in_cluster, key=lambda r: extract_ts(r.get('filename', '')), reverse=True)
+    
+    # 同类型去重:保留最新,记录要删除的文件
+    type_records = defaultdict(list)
+    for r in recs_sorted:
+        rt = r.get('报告类型', '未知')
+        type_records[rt].append(r)
+    
+    deduped = []
+    for rt, recs_list in type_records.items():
+        if len(recs_list) > 1:
+            deduped.append(recs_list[0])  # 保留最新的(已按ts倒序)
+            for old in recs_list[1:]:
+                fname = old.get('filename', '')
+                if fname:
+                    files_to_delete.append(fname)
+        else:
+            deduped.append(recs_list[0])
+    
+    # 构建合并行
+    merged = {
+        '姓名': name,
+        '测评日期': normalize_date(cluster_date) if cluster_date else '',
+        '生日': '',
+        '报告类型汇总': '',
+    }
+    
+    # 收集所有类型 & 生日
+    all_types = set()
+    for r in deduped:
+        rt = r.get('报告类型', '未知')
+        all_types.add(rt)
+        if not merged['生日'] and r.get('生日'):
+            merged['生日'] = r.get('生日')
+    merged['报告类型汇总'] = '+'.join(sorted(all_types))
+    
+    # 为每种报告类型创建总分/百分位字段
+    for rt in ['A1', 'A2', 'B2', 'B3', 'B4', 'B5', 'B6', 'C1']:
+        merged[f'{rt}_总分'] = ''
+        merged[f'{rt}_百分位'] = ''
+    
+    # 按类型归档(已去重,每种最多1条)
+    type_data = {}
+    for r in deduped:
+        rt = r.get('报告类型', '未知')
+        type_data[rt] = r
+    
+    # 核心认知维度 - 优先从A1取, 其次B4, 最后C1
+    cognitive_dims = ['感知觉', '注意力', '记忆力', '推理能力', '空间能力', '加工速度']
+    for type_priority in ['A1', 'B4', 'C1']:
+        if type_priority in type_data:
+            latest = type_data[type_priority]
+            for dim in cognitive_dims:
+                pct_key = f'{dim}_pct'
+                score_key = f'{dim}_score'
+                if pct_key not in merged and latest.get(pct_key):
+                    merged[pct_key] = latest[pct_key]
+                if score_key not in merged and latest.get(score_key):
+                    merged[score_key] = latest[score_key]
+    
+    # 大五人格从C1获取(若无A2)
+    big5_dims = ['开放性', '宜人性', '责任心', '外倾性', '神经质']
+    if 'A2' not in type_data and 'C1' in type_data:
+        c1_data = type_data['C1']
+        for dim in big5_dims:
+            if c1_data.get(dim) and dim not in merged:
+                merged[dim] = c1_data[dim]
+    
+    for rt, rec in type_data.items():
+        if rec.get('总分'):
+            merged[f'{rt}_总分'] = rec.get('总分', '')
+        if rec.get('百分位'):
+            merged[f'{rt}_百分位'] = rec.get('百分位', '')
+        
+        if rt == 'A2':
+            for dim in ['开放性', '宜人性', '外倾性', '神经质', '责任心',
+                       '情绪调节', '抗挫折', '内驱力', '目标管理', '人际交往',
+                       '自卑-自信', '抑郁-安详', '焦虑-安详', '无力感-掌控感',
+                       '情绪总分',
+                       '与父亲信任', '与父亲沟通', '与父亲亲近',
+                       '与母亲信任', '与母亲沟通', '与母亲亲近',
+                       '与同伴信任', '与同伴沟通', '与同伴亲近',
+                       'BMI', '身高', '体重', '睡眠_小时', '饮食_小时']:
+                if rec.get(dim):
+                    merged[dim] = rec.get(dim, '')
+        
+        elif rt == 'B3':
+            for dim in ['抑制控制_pct', '工作记忆_pct', '认知灵活性_pct',
+                       '深层动机', '表面动机', '自我效能感',
+                       '深层方法与策略', '表面方法与策略', '学习自我调节']:
+                if rec.get(dim):
+                    merged[dim] = rec.get(dim, '')
+        
+        elif rt == 'B4':
+            for dim in ['自我概念_行为表现', '自我概念_能力与学校', '自我概念_躯体外貌', 
+                        '自我概念_情绪状态', '自我概念_合群', '自我概念_幸福与满足',
+                        '自主性', '胜任感', '归属感',
+                        '成长型思维']:
+                if rec.get(dim):
+                    merged[dim] = rec.get(dim, '')
+        
+        elif rt == 'B5':
+            for dim in ['认知重评', '表达抑制',
+                       '睡眠_时长', '睡眠_效率', '入睡时间_min', '睡眠_障碍',
+                       '睡眠_主观质量', '睡眠_日间功能', '睡眠_催眠药物',
+                       '网络依赖_总分', '网络依赖_强迫使用', '网络依赖_时间管理', '网络依赖_戒断反应',
+                       '学业压力_学业负担', '学业压力_家庭期望', '学业压力_师生关系',
+                       '学业压力_自我期望', '学业压力_同伴竞争',
+                       '人际_母亲关系', '人际_父亲关系', '人际_师生关系', '人际_同伴关系',
+                       '社交_主动交往', '社交_情感支持', '社交_情感表达', '社交_表达影响', '社交_冲突解决',
+                       '运动_久坐_频率', '运动_久坐_时长',
+                       '运动_步行_频率', '运动_步行_时长',
+                       '运动_中等强度_频率', '运动_中等强度_时长',
+                       '运动_高强度_频率', '运动_高强度_时长']:
+                if rec.get(dim):
+                    merged[dim] = rec.get(dim, '')
+        
+        elif rt == 'B6':
+            for dim in ['兴趣_艺术型', '兴趣_社会型', '兴趣_事业型', '兴趣_常规型', '兴趣_现实型', '兴趣_研究型',
+                       '能力_内省', '能力_空间', '能力_音乐', '能力_人际关系', '能力_自然', '能力_身体运动',
+                       '能力_语言', '能力_逻辑数学', '职业价值观']:
+                if rec.get(dim):
+                    merged[dim] = rec.get(dim, '')
+    
+    merged_students.append(merged)
+
+
+# ===== 第4步:删除过期文件(同类型30天内重复的较早版本) =====
+deleted_count = 0
+for fname in files_to_delete:
+    full = os.path.join(project_dir, '报告', fname)
+    if os.path.exists(full):
+        os.remove(full)
+        subprocess.run(['svn', 'delete', '--non-interactive', full],
+                      capture_output=True, cwd=project_dir)
+        deleted_count += 1
+
+if deleted_count:
+    print(f'删除过期文件: {deleted_count}')
+else:
+    print('无过期文件需删除')
+
+
+# ===== 第5步:定义列顺序并输出 =====
+columns = ['姓名', '测评日期', '生日', '报告类型汇总']
+
+for rt in ['A1', 'A2', 'B2', 'B3', 'B4', 'B5', 'B6', 'C1']:
+    columns.extend([f'{rt}_总分', f'{rt}_百分位'])
+
+cognitive_dims = ['感知觉', '注意力', '记忆力', '推理能力', '空间能力', '加工速度']
+for dim in cognitive_dims:
+    columns.append(f'{dim}_pct')
+for dim in cognitive_dims:
+    columns.append(f'{dim}_score')
+
+# A2 特有
+columns.extend(['开放性', '宜人性', '外倾性', '神经质', '责任心',
+               '情绪调节', '抗挫折', '内驱力', '目标管理', '人际交往',
+               '自卑-自信', '抑郁-安详', '焦虑-安详', '无力感-掌控感',
+               '情绪总分',
+               '与父亲信任', '与父亲沟通', '与父亲亲近',
+               '与母亲信任', '与母亲沟通', '与母亲亲近',
+               '与同伴信任', '与同伴沟通', '与同伴亲近',
+               'BMI', '身高', '体重', '睡眠_小时', '饮食_小时'])
+
+# B3 特有
+columns.extend(['抑制控制_pct', '工作记忆_pct', '认知灵活性_pct',
+               '深层动机', '表面动机', '自我效能感',
+               '深层方法与策略', '表面方法与策略', '学习自我调节'])
+
+# B4 特有
+columns.extend(['自我概念_行为表现', '自我概念_能力与学校', '自我概念_躯体外貌', 
+                '自我概念_情绪状态', '自我概念_合群', '自我概念_幸福与满足',
+                '自主性', '胜任感', '归属感',
+                '成长型思维'])
+
+# B5 特有
+columns.extend(['认知重评', '表达抑制',
+               '睡眠_时长', '睡眠_效率', '入睡时间_min', '睡眠_障碍',
+               '睡眠_主观质量', '睡眠_日间功能', '睡眠_催眠药物',
+               '网络依赖_总分', '网络依赖_强迫使用', '网络依赖_时间管理', '网络依赖_戒断反应',
+               '学业压力_学业负担', '学业压力_家庭期望', '学业压力_师生关系',
+               '学业压力_自我期望', '学业压力_同伴竞争',
+               '人际_母亲关系', '人际_父亲关系', '人际_师生关系', '人际_同伴关系',
+               '社交_主动交往', '社交_情感支持', '社交_情感表达', '社交_表达影响', '社交_冲突解决',
+               '运动_久坐_频率', '运动_久坐_时长',
+               '运动_步行_频率', '运动_步行_时长',
+               '运动_中等强度_频率', '运动_中等强度_时长',
+               '运动_高强度_频率', '运动_高强度_时长'])
+
+# B6 特有
+columns.extend(['兴趣_艺术型', '兴趣_社会型', '兴趣_事业型', '兴趣_常规型', '兴趣_现实型', '兴趣_研究型',
+               '能力_内省', '能力_空间', '能力_音乐', '能力_人际关系', '能力_自然', '能力_身体运动',
+               '能力_语言', '能力_逻辑数学', '职业价值观'])
+
+with open(output_path, 'w', encoding='utf-8-sig', newline='') as f:
+    writer = csv.DictWriter(f, fieldnames=columns)
+    writer.writeheader()
+    writer.writerows(merged_students)
+
+print(f'\n合并后记录数: {len(merged_students)}')
+print(f'输出文件: {output_path}')
+
+# 统计
+type_stats = defaultdict(int)
+for s in merged_students:
+    types = s.get('报告类型汇总', '').split('+')
+    for t in types:
+        if t:
+            type_stats[t] += 1
+
+print(f'\n统计:')
+for t, c in sorted(type_stats.items()):
+    print(f'{t}: {c}')

+ 99 - 0
docs/参考资料/dan_reports/process_all_pdfs.py

@@ -0,0 +1,99 @@
+# -*- coding: utf-8 -*-
+"""
+Process remaining PDFs - analyze content to identify type and add proper prefix
+"""
+import os
+import re
+import fitz
+from type_detector import detect_type
+
+# Known patterns that already have prefix
+ALREADY_PREFIXED = re.compile(r'^(A1_|A2_|B[0-9]_|[A-Z][0-9]\s)')
+
+def identify_report_type(pdf_path):
+    """Identify report type by reading PDF content (fingerprint-based)"""
+    try:
+        doc = fitz.open(pdf_path)
+        text = ""
+        for page in doc[:5]:
+            text += page.get_text()
+        doc.close()
+        
+        report_type, scores = detect_type(text, pdf_path)
+        return report_type
+    except Exception as e:
+        print(f"Error reading {pdf_path}: {e}")
+        return 'Error'
+
+def process_files():
+    # 自动计算项目根目录
+    script_dir = os.path.dirname(os.path.abspath(__file__))
+    project_dir = os.path.dirname(os.path.dirname(script_dir))
+    reports_dir = os.path.join(project_dir, "报告")
+    output_dir = os.path.join(project_dir, "报告")
+    
+    # Get all PDF files
+    all_files = [f for f in os.listdir(reports_dir) if f.endswith('.pdf')]
+    
+    # Filter to unprocessed files (no proper prefix)
+    unprocessed = []
+    for f in all_files:
+        if not ALREADY_PREFIXED.match(f):
+            unprocessed.append(f)
+    
+    print(f"Total files: {len(all_files)}")
+    print(f"Already processed (has prefix): {len(all_files) - len(unprocessed)}")
+    print(f"Need to process: {len(unprocessed)}")
+    
+    # Process in batches
+    results = {'A1': [], 'A2': [], 'B2': [], 'B3': [], 'B4': [], 'B6': [], 'Unknown': [], 'Error': []}
+    
+    for i, filename in enumerate(unprocessed):
+        if i % 50 == 0:
+            print(f"Processing {i}/{len(unprocessed)}...")
+        
+        pdf_path = os.path.join(reports_dir, filename)
+        report_type = identify_report_type(pdf_path)
+        results[report_type].append(filename)
+        
+        # Extract name from filename for renaming
+        name_match = re.search(r'[_-](\w+)[_-]\d{13,}', filename)
+        if name_match:
+            name = name_match.group(1)
+        else:
+            # Try alternative pattern
+            parts = re.split(r'[_\-]', filename)
+            name = parts[-2] if len(parts) >= 2 else 'unknown'
+        
+        # Create new filename with proper prefix
+        ext = '.pdf'
+        # Get name part (before timestamp)
+        name_match = re.search(r'(.+?)_\d{13,}', filename)
+        if name_match:
+            base_name = name_match.group(1).strip()
+            new_filename = f"{report_type}_{base_name}{ext}"
+        else:
+            new_filename = f"{report_type}_{filename}"
+        
+        # Rename if not already correct
+        if not filename.startswith(report_type + '_'):
+            try:
+                new_path = os.path.join(reports_dir, new_filename)
+                # Check if target exists
+                if not os.path.exists(new_path):
+                    os.rename(pdf_path, new_path)
+                    print(f"Renamed: {filename} -> {new_filename}")
+                else:
+                    print(f"Skipped (exists): {filename}")
+            except Exception as e:
+                print(f"Error renaming {filename}: {e}")
+    
+    # Print summary
+    print("\n=== Summary ===")
+    for t, files in results.items():
+        print(f"{t}: {len(files)}")
+    
+    return results
+
+if __name__ == "__main__":
+    process_files()

+ 35 - 0
docs/参考资料/dan_reports/read_pdf.py

@@ -0,0 +1,35 @@
+import pdfplumber
+import sys
+import os
+
+def read_pdf(pdf_path):
+    text = ""
+    try:
+        with pdfplumber.open(pdf_path) as pdf:
+            for page in pdf.pages:
+                text += page.extract_text() or ""
+                text += "\n\n"
+        return text
+    except Exception as e:
+        return f"Error: {e}"
+
+if __name__ == "__main__":
+    if len(sys.argv) < 2:
+        print("Usage: python read_pdf.py <pdf_file>")
+        sys.exit(1)
+    
+    pdf_file = sys.argv[1]
+    if not os.path.exists(pdf_file):
+        print(f"File not found: {pdf_file}")
+        sys.exit(1)
+    
+    content = read_pdf(pdf_file)
+    # 保存到临时文件,避免编码问题
+    import tempfile
+    import os
+    temp_file = os.path.join(tempfile.gettempdir(), 'pdf_content.txt')
+    with open(temp_file, 'w', encoding='utf-8') as f:
+        f.write(content)
+    print(f"Content saved to: {temp_file}")
+    # 只打印关键信息
+    print(content[:500] if content else "No content extracted")

+ 225 - 0
docs/参考资料/dan_reports/type_detector.py

@@ -0,0 +1,225 @@
+# -*- coding: utf-8 -*-
+"""
+报告类型指纹检测模块
+
+基于 PDF 文本内容的加权指纹评分系统,用于识别报告类型。
+被 process_all_pdfs.py 和 extract_all_types.py 共享使用。
+
+用法:
+    from type_detector import detect_type
+    report_type, scores = detect_type(text, filename)
+
+指纹按唯一性从高到低排列。排除规则用于处理有冲突特征的情况(如 B4 含"校园版"则扣分)。
+阈值保护防止对文本层损坏的 PDF 误判,安全回退到文件名前缀匹配。
+"""
+
+# 类型指纹注册表:每个条目 (文本特征, 权重)
+# 多个指纹之和为该类型的总分
+FINGERPRINTS = {
+    'B5': [
+        ('青春期挑战', 25),     # 独占标题
+        ('学业压力源', 20),     # 独占章节
+        ('网络依赖', 18),       # 独占章节
+        ('社交能力', 8),        # 辅助特征
+    ],
+    'B4': [
+        ('GROWTH MINDSET', 30),          # 英文独占,提取最稳定
+        ('成长型思维', 25),              # 中文独占
+        ('SELF-CONCEPT REPORT', 10),     # 与B2共享,权重较低
+        ('核心认知能力', 8),             # B4标识,区别于A1/C1
+        ('认知能力测评报告', 5),         # 共享(B4含认知模块,低权重以免抢A1)
+    ],
+    'B6': [
+        ('CAREER', 25),                  # 独占英文标题
+        ('职业发展', 22),                # 独占中文标题
+        ('Holland', 20),                 # 霍兰德职业代码
+        ('多元智能', 10),                # 辅助特征
+    ],
+    'B2': [
+        ('CHILD BEHAVIOR', 25),          # 独占:儿童行为测评报告
+        ('Conduct problems', 20),        # 独占:品行问题英文名
+        ('PARENTAL EDUCATION', 20),      # 独占:父母教养测评报告
+        ('FAMILY ENVIRONMENT', 18),      # 独占:家庭教育环境测评报告
+        ('家庭教养', 12),                # 辅助特征
+    ],
+    'A2': [
+        ('大五人格', 22),                # 独占标题
+        ('核心素养', 20),                # 独占标题
+        ('EMOTION', 12),                 # 辅助特征(情绪量表区域)
+        ('Big Five', 10),               # 英文辅助
+        ('综合发展', 8),                 # A2报告常见标题
+    ],
+    'B3': [
+        ('MOTIVATION', 22),              # 独占英文标题
+        ('学习动机', 20),                # 独占中文标题
+        ('深层动机', 12),                # 辅助特征
+    ],
+    'A1': [
+        ('认知能力测评报告', 15),        # 共享标题(A1/C1/B4都有)
+        ('COGNITIVE ABILITY REPORT', 15),# 共享英文标题
+    ],
+    'C1': [
+        ('校园版', 35),                  # 独占标识(提高权重以胜出共享指纹)
+        ('认知能力测评报告', 5),         # 共享,低权重
+        ('COGNITIVE ABILITY REPORT', 5), # 共享,低权重
+    ],
+}
+
+# 排除规则:冲突文本 → 从特定类型扣分
+# 当某类型有不应含有的特征时,用负权重拉低其总分
+EXCLUSIONS = {
+    'A1': [('GROWTH MINDSET', -20)],        # A1 不含成长型思维(轻扣分,避免抵消文件名加成)
+    'C1': [('GROWTH MINDSET', -50)],        # C1 不含成长型思维
+    'B4': [('校园版', -50), ('PARENTAL EDUCATION', -40)],  # B4 非校园版,无教养内容
+    'A2': [('Conduct problems', -40)],      # A2 无儿童行为
+}
+
+# 阈值:低于此分数时回退到文件名匹配
+THRESHOLD = 10
+
+
+def detect_type(text, filename=''):
+    """基于PDF文本内容和文件名检测报告类型(混合评分)。
+    
+    先用内容指纹评分,再叠加文件名信号作为强补充。
+    当文件名明确指示类型时,该类型获得+50加成(解决C1/A1缺文本指纹的问题)。
+    
+    Args:
+        text: PDF 全文文本
+        filename: PDF 文件名(用于回退和混合评分)
+    
+    Returns:
+        (report_type: str, scores: dict) — 类型和得分详情
+    """
+    if not text or not text.strip():
+        return fallback_by_filename(filename), {}
+    
+    # 初始化分数
+    scores = {t: 0 for t in FINGERPRINTS}
+    
+    # 1. 正向评分:每种类型匹配其指纹
+    for report_type, patterns in FINGERPRINTS.items():
+        for pattern, weight in patterns:
+            if pattern in text:
+                scores[report_type] += weight
+    
+    # 2. 排除扣分:冲突特征拉低特定类型分数
+    for report_type, rules in EXCLUSIONS.items():
+        for pattern, penalty in rules:
+            if pattern in text:
+                scores[report_type] += penalty  # penalty 为负值
+    
+    # 3. 文件名信号加成:当文件名明确指示该类型时,大幅加分
+    #    解决C1/A1等类型缺文本层唯一指纹的问题
+    content_scores = scores.copy()
+    fb_type = fallback_by_filename(filename)
+    if fb_type in scores:
+        scores[fb_type] += 50  # 文件名信号:强于大多数内容指纹
+    
+    # 4. 取最高分
+    top_type = max(scores, key=scores.get)
+    top_score = content_scores[top_type]
+    
+    # 5. 阈值保护:纯内容分数过低则回退到文件名匹配
+    if top_score < THRESHOLD:
+        return fallback_by_filename(filename), scores
+    
+    return top_type, scores
+
+
+def fallback_by_filename(filename):
+    """根据文件名前缀判断报告类型(后备方案)。
+    
+    当内容指纹得分低于阈值时使用此方法。
+    保持与旧版 determine_report_type() 相同的逻辑。
+    """
+    name = filename.upper()
+    # B5 在 A2 之前检查(A2_B5 前缀文件应归为 B5)
+    if 'B5' in name:
+        return 'B5'
+    if 'A1' in name:
+        return 'A1'
+    if 'A2' in name:
+        return 'A2'
+    if 'B2' in name:
+        return 'B2'
+    if 'B3' in name:
+        return 'B3'
+    if 'B4' in name or '综合素质' in name or '成长型思维' in name or '多元认知' in name:
+        return 'B4'
+    if 'B6' in name:
+        return 'B6'
+    if 'C1' in name:
+        return 'C1'
+    return 'Unknown'
+
+
+def debug_detect(text, filename=''):
+    """调试版本,返回详细匹配信息。
+    
+    Returns:
+        (report_type, scores, matches)
+        matches: [(type, pattern, weight), ...] 每项匹配
+    """
+    if not text or not text.strip():
+        return fallback_by_filename(filename), {}, []
+    
+    scores = {t: 0 for t in FINGERPRINTS}
+    matches = []
+    
+    for report_type, patterns in FINGERPRINTS.items():
+        for pattern, weight in patterns:
+            if pattern in text:
+                scores[report_type] += weight
+                matches.append((report_type, pattern, weight))
+    
+    for report_type, rules in EXCLUSIONS.items():
+        for pattern, penalty in rules:
+            if pattern in text:
+                scores[report_type] += penalty
+                matches.append((report_type, f'EXCLUDE:{pattern}', penalty))
+    
+    # 文件名信号
+    content_scores = scores.copy()
+    fb_type = fallback_by_filename(filename)
+    if fb_type in scores:
+        scores[fb_type] += 50
+        matches.append((fb_type, 'FILENAME_BONUS', 50))
+    
+    top_type = max(scores, key=scores.get)
+    top_score = content_scores[top_type]
+    
+    if top_score < THRESHOLD:
+        fb = fallback_by_filename(filename)
+        return fb, scores, matches
+    
+    return top_type, scores, matches
+
+
+if __name__ == '__main__':
+    # 简单自测
+    test_texts = [
+        ('认知能力测评报告', 'A1'),                     # 纯认知报告 = A1
+        ('核心素养与大五人格EMOTION', 'A2'),           # A2 独占指纹
+        ('CHILD BEHAVIOR Conduct problems PARENTING', 'B2'),
+        ('MOTIVATION学习动机深层动机', 'B3'),
+        ('GROWTH MINDSET成长型思维\n认知能力测评报告', 'B4'),  # B4独占 > 共享
+        ('青春期挑战学业压力源网络依赖', 'B5'),
+        ('CAREER职业发展Holland多元智能', 'B6'),
+        ('认知能力测评报告\nCOGNITIVE ABILITY REPORT\n校园版', 'C1'),  # 校园版 > A1
+    ]
+    
+    print('=== Type Detector Self-Test ===')
+    for text, expected in test_texts:
+        result, scores = detect_type(text)
+        ok = 'OK' if result == expected else 'FAIL'
+        print(f'  {ok:4s} {result:4s} (expected {expected}) scores={scores}')
+    
+    # 边界测试
+    print('\n=== Edge Cases ===')
+    print(f'  empty: {detect_type("")[0]} (expected Unknown)')
+    print(f'  only campus: {detect_type("校园版")[0]} (expected C1)')
+    print(f'  A2_B5: {detect_type("青春期挑战", "A2_B5_张欣怡.pdf")[0]} (expected B5)')
+    print(f'  threshold: {detect_type("普通文本")[0]} (expected Unknown)')
+    print(f'  cognitive+校园版: {detect_type("认知能力测评报告COGNITIVE ABILITY REPORT校园版")[0]} (expected C1)')
+    print(f'  B4+校园版: {detect_type("GROWTH MINDSET成长型思维校园版")[0]} (expected Unknown, 校园版扣分后B4<阈值→回退)')

+ 74 - 0
docs/参考资料/dan_reports/update_materials.py

@@ -0,0 +1,74 @@
+# -*- coding: utf-8 -*-
+"""
+自动化更新脚本 - 素材库全量更新流水线
+
+功能: 从原始PDF报告更新素材库的三个核心数据文件
+流程:
+  1. 分类重命名报告 (process_all_pdfs.py)
+  2. 全类型数据提取 (extract_all_types.py)
+  3. 全类型学生合并 (merge_all_types.py)
+  4. 删除重复A1报告 (delete_duplicate_a1.py)
+
+用法:
+  python scripts/reports/update_materials.py          # 从 scripts 目录运行
+  cd 项目根目录 && python scripts/reports/update_materials.py  # 从根目录运行
+"""
+import os
+import sys
+import subprocess
+
+# 脚本所在目录 = scripts/reports/
+SCRIPTS_DIR = os.path.dirname(os.path.abspath(__file__))
+PROJECT_DIR = os.path.dirname(os.path.dirname(SCRIPTS_DIR))
+
+
+def run_script(script_name, label):
+    """运行 scripts/reports/ 中的脚本"""
+    script_path = os.path.join(SCRIPTS_DIR, script_name)
+    print(f"\n{'=' * 50}")
+    print(f"[{label}] 运行: {script_name}")
+    print(f"{'=' * 50}")
+
+    result = subprocess.run(
+        [sys.executable, script_path],
+        cwd=PROJECT_DIR,
+        capture_output=False
+    )
+
+    if result.returncode != 0:
+        print(f"  [FAIL] [{label}] 失败 (exit code={result.returncode})")
+        return False
+    print(f"  [OK] [{label}] 完成")
+    return True
+
+
+def main():
+    print("=" * 60)
+    print("  素材库全量更新流水线")
+    print("=" * 60)
+    print(f"  项目根目录: {PROJECT_DIR}")
+    print(f"  脚本目录:   {SCRIPTS_DIR}")
+
+    steps = [
+        ("process_all_pdfs.py",       "1/4 报告分类重命名"),
+        ("extract_all_types.py",      "2/4 全类型数据提取"),
+        ("merge_all_types.py",        "3/4 全类型学生合并"),
+        ("delete_duplicate_a1.py",    "4/4 清理重复A1报告"),
+    ]
+
+    for script, label in steps:
+        if not run_script(script, label):
+            print(f"\n  ❌ 流水线在 [{label}] 中断")
+            sys.exit(1)
+
+    print(f"\n{'=' * 60}")
+    print("  [OK] 素材库更新完成!")
+    print(f"{'=' * 60}")
+    print("\n生成的文件:")
+    print("  - 素材库/测评数据_优化提取.csv")
+    print("  - 素材库/学生汇总数据.csv")
+    print("\n提示: 案例库.md 需要手动根据新增数据更新案例")
+
+
+if __name__ == "__main__":
+    main()