|
@@ -0,0 +1,1266 @@
|
|
|
|
|
+# -*- coding: utf-8 -*-
|
|
|
|
|
+"""
|
|
|
|
|
+完善素材库 - 处理所有报告类型
|
|
|
|
|
+提取A1, A2, B2, B3, B4, B6, C1报告数据
|
|
|
|
|
+"""
|
|
|
|
|
+import fitz
|
|
|
|
|
+import re
|
|
|
|
|
+import os
|
|
|
|
|
+import csv
|
|
|
|
|
+from type_detector import detect_type
|
|
|
|
|
+import math
|
|
|
|
|
+import io
|
|
|
|
|
+from PIL import Image
|
|
|
|
|
+
|
|
|
|
|
+def extract_text_from_pdf(pdf_path, return_pages=False):
|
|
|
|
|
+ """使用PyMuPDF提取文本"""
|
|
|
|
|
+ try:
|
|
|
|
|
+ doc = fitz.open(pdf_path)
|
|
|
|
|
+
|
|
|
|
|
+ if return_pages:
|
|
|
|
|
+ page_texts = []
|
|
|
|
|
+ for page in doc:
|
|
|
|
|
+ page_texts.append(page.get_text())
|
|
|
|
|
+ doc.close()
|
|
|
|
|
+ return page_texts
|
|
|
|
|
+
|
|
|
|
|
+ text = ""
|
|
|
|
|
+ for page in doc:
|
|
|
|
|
+ text += page.get_text()
|
|
|
|
|
+ doc.close()
|
|
|
|
|
+ return text
|
|
|
|
|
+ except Exception as e:
|
|
|
|
|
+ print(f"Error extracting text from {pdf_path}: {e}")
|
|
|
|
|
+ return "" if not return_pages else []
|
|
|
|
|
+
|
|
|
|
|
+def extract_name_from_filename(filename):
|
|
|
|
|
+ """从文件名提取姓名 - 改进版"""
|
|
|
|
|
+ name = filename.replace('.pdf', '')
|
|
|
|
|
+
|
|
|
|
|
+ # 优先尝试从文件名中提取姓名
|
|
|
|
|
+ # 注意:只用下划线分割,不用连字符(-),避免把"(1-3年级)"切碎
|
|
|
|
|
+ parts = re.split(r'_', name)
|
|
|
|
|
+
|
|
|
|
|
+ # 先找包含中文的部分(姓名通常含中文)
|
|
|
|
|
+ chinese_parts = []
|
|
|
|
|
+ for p in parts:
|
|
|
|
|
+ if re.search(r'[\u4e00-\u9fff]{2,}', p):
|
|
|
|
|
+ # 排除包含关键词的
|
|
|
|
|
+ keywords = ['儿童', '核心', '认知', '发展', 'DAN', '测评', '自我', '综合',
|
|
|
|
|
+ '家庭', '职业', '校园', '准备', '素养', '人格', '思维', '能力',
|
|
|
|
|
+ '青少年', '素质', '成长', '学习', '动机', '兴趣', '规划',
|
|
|
|
|
+ '认知能力', '成长型', '多元', '智能', '教养', '亲子', '欺凌',
|
|
|
|
|
+ '心理', '教育', '教师', '反馈', '报告', '班级', '校级',
|
|
|
|
|
+ '加工速度', '注意力', '记忆力', '推理', '空间', '知觉',
|
|
|
|
|
+ 'B5', 'B6', 'C1', '标准版', '专业版', '高中段']
|
|
|
|
|
+ p_clean = re.sub(r'[A-Za-z0-9\-()]', '', p) # 去掉字母数字和连字符/括号
|
|
|
|
|
+ if p_clean and not any(kw in p for kw in keywords):
|
|
|
|
|
+ chinese_parts.append(p_clean)
|
|
|
|
|
+
|
|
|
|
|
+ if chinese_parts:
|
|
|
|
|
+ # 取第一个/最长的中文名;等长时优先取不含特殊字符的
|
|
|
|
|
+ max_len = max(len(c) for c in chinese_parts)
|
|
|
|
|
+ candidates = [c for c in chinese_parts if len(c) == max_len]
|
|
|
|
|
+ # 等长时取最不像描述文本的(包含 级/挑战/版/段 这类字眼的靠后)
|
|
|
|
|
+ if len(candidates) > 1:
|
|
|
|
|
+ bad_markers = ['级', '挑战', '版', '段', '测', '试', '报', '告']
|
|
|
|
|
+ def badness(s):
|
|
|
|
|
+ return sum(1 for ch in bad_markers if ch in s)
|
|
|
|
|
+ candidates.sort(key=lambda s: (badness(s), chinese_parts.index(s)))
|
|
|
|
|
+ return candidates[0]
|
|
|
|
|
+
|
|
|
|
|
+ # 备选1: 对含连字符(-)的part再按-拆分(兼容旧版DAN格式如"素质能力测评-小明2024-10-069193")
|
|
|
|
|
+ sub_chinese = []
|
|
|
|
|
+ for p in parts:
|
|
|
|
|
+ if '-' not in p:
|
|
|
|
|
+ continue
|
|
|
|
|
+ sub_parts = re.split(r'-', p)
|
|
|
|
|
+ for sp in sub_parts:
|
|
|
|
|
+ if re.search(r'[\u4e00-\u9fff]{2,}', sp):
|
|
|
|
|
+ sp_clean = re.sub(r'[A-Za-z0-9\-()]', '', sp)
|
|
|
|
|
+ if sp_clean and not any(kw in sp for kw in keywords):
|
|
|
|
|
+ sub_chinese.append(sp_clean)
|
|
|
|
|
+ if sub_chinese:
|
|
|
|
|
+ max_len = max(len(c) for c in sub_chinese)
|
|
|
|
|
+ candidates = [c for c in sub_chinese if len(c) == max_len]
|
|
|
|
|
+ if len(candidates) > 1:
|
|
|
|
|
+ bad_markers = ['级', '挑战', '版', '段', '测', '试', '报', '告']
|
|
|
|
|
+ candidates.sort(key=lambda s: (sum(1 for ch in bad_markers if ch in s), sub_chinese.index(s)))
|
|
|
|
|
+ return candidates[0]
|
|
|
|
|
+
|
|
|
|
|
+ # 备选2: 按关键词过滤整块part
|
|
|
|
|
+ keywords = ['儿童', '核心', '认知', '发展', 'A1', 'A2', 'B2', 'B3', 'B4', 'B5', 'B6',
|
|
|
|
|
+ 'DAN', '测评', '自我', '综合', '家庭', '职业', '校园', '准备', '素养',
|
|
|
|
|
+ '人格', '思维', '能力', '青少年', '素质', '成长', '学习', '动机', '兴趣',
|
|
|
|
|
+ '规划', '加工速度', '注意力', '记忆力', '推理', '空间', '知觉']
|
|
|
|
|
+ for p in parts:
|
|
|
|
|
+ if not p:
|
|
|
|
|
+ continue
|
|
|
|
|
+ if re.match(r'^\d{8,}$', p):
|
|
|
|
|
+ continue
|
|
|
|
|
+ if any(kw == p or (p.startswith(kw) and len(p) > len(kw)) for kw in keywords):
|
|
|
|
|
+ continue
|
|
|
|
|
+ if any(kw in p for kw in keywords):
|
|
|
|
|
+ continue
|
|
|
|
|
+ if len(p) >= 2 and not re.match(r'^\d+$', p):
|
|
|
|
|
+ return p.strip()
|
|
|
|
|
+ return parts[0] if parts else name
|
|
|
|
|
+
|
|
|
|
|
+def extract_birthday_from_pdf(text, page_texts=None):
|
|
|
|
|
+ """从PDF中提取生日"""
|
|
|
|
|
+ if page_texts and len(page_texts) > 1:
|
|
|
|
|
+ page2_text = page_texts[1]
|
|
|
|
|
+ else:
|
|
|
|
|
+ page2_text = text
|
|
|
|
|
+
|
|
|
|
|
+ patterns = [
|
|
|
|
|
+ r'出生日期[::\s]*(\d{4}[-/年]\d{1,2}[-/月]\d{1,2})',
|
|
|
|
|
+ r'(\d{4}[-/年]\d{1,2}[-/月]\d{1,2})',
|
|
|
|
|
+ ]
|
|
|
|
|
+
|
|
|
|
|
+ for pattern in patterns:
|
|
|
|
|
+ match = re.search(pattern, page2_text)
|
|
|
|
|
+ if match:
|
|
|
|
|
+ date_str = match.group(1)
|
|
|
|
|
+ date_str = date_str.replace('年', '-').replace('月', '-').replace('/', '-')
|
|
|
|
|
+ if re.match(r'\d{4}-\d{2}-\d{2}', date_str):
|
|
|
|
|
+ return date_str
|
|
|
|
|
+ return ""
|
|
|
|
|
+
|
|
|
|
|
+def extract_score_and_percentile(text):
|
|
|
|
|
+ """通用提取总分和百分位(所有报告类型共用)
|
|
|
|
|
+
|
|
|
|
|
+ PDF格式: "112\n总得分\nTotal Score\n79\n百分位(%)"
|
|
|
|
|
+ 总分 = "总得分"前面的数字, 百分位 = "总得分"和"百分位"之间的数字
|
|
|
|
|
+ """
|
|
|
|
|
+ result = {}
|
|
|
|
|
+
|
|
|
|
|
+ # 主格式: 数字\n总得分 (所有类型报告通用)
|
|
|
|
|
+ score_match = re.search(r'(\d+)\s*\n\s*总得分', text)
|
|
|
|
|
+ if score_match:
|
|
|
|
|
+ result['总分'] = score_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 百分位: 总得分...数字...百分位
|
|
|
|
|
+ pct_match = re.search(r'总得分.*?(\d+)\s*百分位', text, re.DOTALL)
|
|
|
|
|
+ if pct_match:
|
|
|
|
|
+ result['百分位'] = pct_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 后备1: 旧版 "总分" 格式
|
|
|
|
|
+ if '总分' not in result:
|
|
|
|
|
+ total_match = re.search(r'总分[^\d]*(\d+)', text)
|
|
|
|
|
+ if total_match:
|
|
|
|
|
+ result['总分'] = total_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 后备2: 旧版 "百分位" 格式
|
|
|
|
|
+ if '百分位' not in result:
|
|
|
|
|
+ pct_match2 = re.search(r'百分位[^\d]*(\d+)', text)
|
|
|
|
|
+ if pct_match2:
|
|
|
|
|
+ result['百分位'] = pct_match2.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ return result
|
|
|
|
|
+
|
|
|
|
|
+def extract_a1_data(text, page_texts=None):
|
|
|
|
|
+ """提取A1报告数据 - 儿童核心认知发展(6个认知维度)"""
|
|
|
|
|
+ result = {}
|
|
|
|
|
+
|
|
|
|
|
+ result.update(extract_score_and_percentile(text))
|
|
|
|
|
+
|
|
|
|
|
+ # 6个核心认知维度: 感知觉, 注意力, 记忆力, 推理能力, 空间能力, 加工速度
|
|
|
|
|
+ cognitive_dims = ['感知觉', '注意力', '记忆力', '推理能力', '空间能力', '加工速度']
|
|
|
|
|
+
|
|
|
|
|
+ # 方法1: 从summary页面提取百分位 (格式: "感知觉 | Perception\n百分位(%)\n14")
|
|
|
|
|
+ for dim in cognitive_dims:
|
|
|
|
|
+ pct_match = re.search(
|
|
|
|
|
+ rf'{re.escape(dim)}\s*\|\s*\w+\s*\n\s*百分位(%)\s*\n\s*(\d+)',
|
|
|
|
|
+ text
|
|
|
|
|
+ )
|
|
|
|
|
+ if pct_match:
|
|
|
|
|
+ result[f'{dim}_pct'] = pct_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 方法2: 从detail页面提取原始分和百分位 (格式: "我的感知觉得分 41 | 14%")
|
|
|
|
|
+ for dim in cognitive_dims:
|
|
|
|
|
+ detail_match = re.search(
|
|
|
|
|
+ rf'我的{re.escape(dim)}得分\s*(\d+)\s*\|\s*(\d+)%',
|
|
|
|
|
+ text
|
|
|
|
|
+ )
|
|
|
|
|
+ if detail_match:
|
|
|
|
|
+ result[f'{dim}_score'] = detail_match.group(1)
|
|
|
|
|
+ if f'{dim}_pct' not in result:
|
|
|
|
|
+ result[f'{dim}_pct'] = detail_match.group(2)
|
|
|
|
|
+
|
|
|
|
|
+ # 方法3: 后备 - 直接搜索 "维度名\n百分位(%)\n数字" (某些PDF格式)
|
|
|
|
|
+ if len([k for k in result if k.endswith('_pct')]) < 6:
|
|
|
|
|
+ for dim in cognitive_dims:
|
|
|
|
|
+ if f'{dim}_pct' not in result:
|
|
|
|
|
+ fallback = re.search(
|
|
|
|
|
+ rf'{re.escape(dim)}.*?百分位(%).*?(\d+)',
|
|
|
|
|
+ text,
|
|
|
|
|
+ re.DOTALL
|
|
|
|
|
+ )
|
|
|
|
|
+ if fallback:
|
|
|
|
|
+ result[f'{dim}_pct'] = fallback.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ return result
|
|
|
|
|
+
|
|
|
|
|
+def extract_a2_data(text, page_texts=None):
|
|
|
|
|
+ """提取A2报告数据 - 核心素养16+(认知+人格+情绪+关系+健康)"""
|
|
|
|
|
+ result = {}
|
|
|
|
|
+
|
|
|
|
|
+ # 总分和百分位 (通用格式)
|
|
|
|
|
+ result.update(extract_score_and_percentile(text))
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 认知维度 (3个: 感知觉, 记忆力, 注意力) =====
|
|
|
|
|
+ # 在summary页面中: "感知觉\n89%"
|
|
|
|
|
+ for dim in ['感知觉', '记忆力', '注意力']:
|
|
|
|
|
+ cog_match = re.search(rf'{re.escape(dim)}\s*\n\s*(\d+)%', text)
|
|
|
|
|
+ if cog_match:
|
|
|
|
|
+ result[f'{dim}_pct'] = cog_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 情绪状态 (第5页/索引4) =====
|
|
|
|
|
+ if page_texts and len(page_texts) > 4:
|
|
|
|
|
+ page5_text = page_texts[4]
|
|
|
|
|
+
|
|
|
|
|
+ # 方法1: 原始格式 "数字\nInferiority"
|
|
|
|
|
+ emotion_names = ['自卑-自信', '抑郁-安详', '焦虑-安详', '无力感-掌控感']
|
|
|
|
|
+ emotion_eng = ['Inferiority', 'Depression', 'Anxiety', 'Helpless']
|
|
|
|
|
+
|
|
|
|
|
+ for eng, cn in zip(emotion_eng, emotion_names):
|
|
|
|
|
+ em_match = re.search(rf'(\d+(?:\.?\d+)?)\s*\n\s*{re.escape(eng)}', page5_text)
|
|
|
|
|
+ if em_match:
|
|
|
|
|
+ result[cn] = em_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 方法2: 旧格式 (后备)
|
|
|
|
|
+ if not any(k in result for k in emotion_names):
|
|
|
|
|
+ emotion_pattern = re.findall(r'(\d+)\s+10\s+(\d+\.?\d*)\s+(\d+\.?\d*)', page5_text)
|
|
|
|
|
+ if len(emotion_pattern) >= 1:
|
|
|
|
|
+ for i in range(min(len(emotion_pattern), 4)):
|
|
|
|
|
+ result[emotion_names[i]] = emotion_pattern[i][-1]
|
|
|
|
|
+
|
|
|
|
|
+ # 情绪总分: "在4个分测验中的总得分是 XX分"
|
|
|
|
|
+ total_em = re.search(r'在4个分测验中的总得分是\s*(\d+(?:\.?\d+)?)分', page5_text)
|
|
|
|
|
+ if total_em:
|
|
|
|
|
+ result['情绪总分'] = total_em.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 大五人格 (第7页/索引6) =====
|
|
|
|
|
+ if page_texts and len(page_texts) > 6:
|
|
|
|
|
+ page7_text = page_texts[6]
|
|
|
|
|
+
|
|
|
|
|
+ # 方法1: 直接匹配 "您在"XX"上的得分是 X.X 分"
|
|
|
|
|
+ big5_names = ['开放性', '宜人性', '责任心', '外倾性', '神经质']
|
|
|
|
|
+ for dim in big5_names:
|
|
|
|
|
+ b5_match = re.search(rf'您在[""「]{re.escape(dim)}[""」].*?得分是\s*(\d+\.?\d*)\s*分', page7_text)
|
|
|
|
|
+ if b5_match:
|
|
|
|
|
+ result[dim] = b5_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 方法2: 小数点回退 (旧方法)
|
|
|
|
|
+ if len([k for k in result if k in big5_names]) < 5:
|
|
|
|
|
+ decimal_pattern = re.findall(r'\b(\d+\.\d+)\b', page7_text)
|
|
|
|
|
+ valid_decimals = [d for d in decimal_pattern if 1.0 <= float(d) <= 10.0]
|
|
|
|
|
+ for i, dim_name in enumerate(big5_names):
|
|
|
|
|
+ if i < len(valid_decimals) and dim_name not in result:
|
|
|
|
|
+ result[dim_name] = valid_decimals[i]
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 社会关系 (第9页/索引8) =====
|
|
|
|
|
+ if page_texts and len(page_texts) > 8:
|
|
|
|
|
+ page9_text = page_texts[8]
|
|
|
|
|
+ all_num_matches = re.findall(r'(\d+)\s*分', page9_text)
|
|
|
|
|
+ valid_nums = [n for n in all_num_matches if 1 <= int(n) <= 100]
|
|
|
|
|
+
|
|
|
|
|
+ if len(valid_nums) >= 9:
|
|
|
|
|
+ # 页面9段落中数字按子维度分组(信赖→沟通→亲近), 每组内按母亲→父亲→同伴排列
|
|
|
|
|
+ # "信赖...37分,40分,34分;沟通上...32分,32分,25分;亲近上...15分,16分,17分"
|
|
|
|
|
+ result['与母亲信任'] = valid_nums[0]
|
|
|
|
|
+ result['与父亲信任'] = valid_nums[1]
|
|
|
|
|
+ result['与同伴信任'] = valid_nums[2]
|
|
|
|
|
+ result['与母亲沟通'] = valid_nums[3]
|
|
|
|
|
+ result['与父亲沟通'] = valid_nums[4]
|
|
|
|
|
+ result['与同伴沟通'] = valid_nums[5]
|
|
|
|
|
+ result['与母亲亲近'] = valid_nums[6]
|
|
|
|
|
+ result['与父亲亲近'] = valid_nums[7]
|
|
|
|
|
+ result['与同伴亲近'] = valid_nums[8]
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 身体健康 (第11页/索引10) =====
|
|
|
|
|
+ # 格式: "BMI:22kg/m²" "身高:156cm" "体重:53kg" "10小时/周" "9小时/每天"
|
|
|
|
|
+ if page_texts and len(page_texts) > 10:
|
|
|
|
|
+ page11_text = page_texts[10]
|
|
|
|
|
+
|
|
|
|
|
+ # BMI格式: "BMI:22kg/m²"
|
|
|
|
|
+ bmi_match = re.search(r'BMI[::]*\s*(\d+(?:\.\d+)?)\s*kg', page11_text)
|
|
|
|
|
+ if bmi_match:
|
|
|
|
|
+ result['BMI'] = bmi_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 身高: "身高:156cm"
|
|
|
|
|
+ height_match = re.search(r'身高[::]*\s*(\d+)\s*cm', page11_text)
|
|
|
|
|
+ if height_match:
|
|
|
|
|
+ result['身高'] = height_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 体重: "体重:53kg"
|
|
|
|
|
+ weight_match = re.search(r'体重[::]*\s*(\d+(?:\.\d+)?)\s*kg', page11_text)
|
|
|
|
|
+ if weight_match:
|
|
|
|
|
+ result['体重'] = weight_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 睡眠: "10小时/周" 或 "9小时/每天" (先找周再找天)
|
|
|
|
|
+ sleep_week = re.search(r'(\d+)\s*小时\s*/\s*周', page11_text)
|
|
|
|
|
+ if sleep_week:
|
|
|
|
|
+ result['睡眠_小时'] = sleep_week.group(1)
|
|
|
|
|
+ else:
|
|
|
|
|
+ sleep_day = re.search(r'(\d+)\s*小时\s*/\s*每天', page11_text)
|
|
|
|
|
+ if sleep_day:
|
|
|
|
|
+ result['睡眠_小时'] = sleep_day.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 饮食: "9小时/每天"
|
|
|
|
|
+ diet = re.search(r'(\d+)\s*小时\s*/\s*每天', page11_text)
|
|
|
|
|
+ if diet and '饮食' not in result:
|
|
|
|
|
+ result['饮食_小时'] = diet.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 运动: 找"运动习惯"附近的"X小时/周"
|
|
|
|
|
+ exercise = re.search(r'运动.*?(\d+)\s*小时\s*/\s*周', page11_text, re.DOTALL)
|
|
|
|
|
+ if exercise:
|
|
|
|
|
+ result['运动_小时'] = exercise.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ return result
|
|
|
|
|
+
|
|
|
|
|
+def extract_b2_data(text, page_texts=None):
|
|
|
|
|
+ """提取B2报告数据 - 儿童自我与家庭教养"""
|
|
|
|
|
+ result = {}
|
|
|
|
|
+ result.update(extract_score_and_percentile(text))
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 自我概念 (6维度) =====
|
|
|
|
|
+ # 格式: "行为表现 9分 能力与学校表现 8分 躯体外貌 9分..."
|
|
|
|
|
+ # 注意: 实际PDF中维度的准确名称
|
|
|
|
|
+ self_concept_dims = [
|
|
|
|
|
+ ('行为表现', '行为表现'),
|
|
|
|
|
+ ('能力与学校表现', '能力与学校'), # PDF用"能力与学校表现"
|
|
|
|
|
+ ('躯体外貌', '躯体外貌'),
|
|
|
|
|
+ ('情绪状态', '情绪状态'),
|
|
|
|
|
+ ('合群', '合群'),
|
|
|
|
|
+ ('幸福与满足', '幸福与满足'),
|
|
|
|
|
+ ]
|
|
|
|
|
+ for dim_pdf, dim_out in self_concept_dims:
|
|
|
|
|
+ sc_match = re.search(rf'{re.escape(dim_pdf)}\s*(\d+)\s*分', text)
|
|
|
|
|
+ if sc_match:
|
|
|
|
|
+ result[f'自我概念_{dim_out}'] = sc_match.group(1)
|
|
|
|
|
+ else:
|
|
|
|
|
+ # 后备: 包含匹配
|
|
|
|
|
+ sc_match2 = re.search(rf'{re.escape(dim_pdf)}.*?(\d+)\s*分', text[:2000])
|
|
|
|
|
+ if sc_match2:
|
|
|
|
|
+ result[f'自我概念_{dim_out}'] = sc_match2.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 儿童行为 (8维度) =====
|
|
|
|
|
+ # 格式: "Conduct problems 1分 优秀 情绪问题 Emotional state 2分 良好..."
|
|
|
|
|
+ behavior_dims_cn = ['品行问题', '情绪问题', '学习问题', '社交问题', '生活习惯', '多动倾向', '刻板行为', '拖延行为']
|
|
|
|
|
+ for dim in behavior_dims_cn:
|
|
|
|
|
+ b_match = re.search(rf'{re.escape(dim)}\s*(\d+)\s*分', text)
|
|
|
|
|
+ if b_match:
|
|
|
|
|
+ result[f'行为_{dim}'] = b_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 备选: 英文关键词
|
|
|
|
|
+ behavior_eng = [
|
|
|
|
|
+ (r'Conduct problems\s+(\d+)分', '品行问题'),
|
|
|
|
|
+ (r'Emotional state\s+(\d+)分', '情绪问题'),
|
|
|
|
|
+ (r'Learning situation\s+(\d+)分', '学习问题'),
|
|
|
|
|
+ (r'Social situation\s+(\d+)分', '社交问题'),
|
|
|
|
|
+ (r'Habits.*?customs\s+(\d+)分', '生活习惯'),
|
|
|
|
|
+ (r'Hyperactivity\s+(\d+)分', '多动倾向'),
|
|
|
|
|
+ (r'Stereotypic\s+(\d+)分', '刻板行为'),
|
|
|
|
|
+ (r'Procrastination\s+(\d+)分', '拖延行为'),
|
|
|
|
|
+ ]
|
|
|
|
|
+ for eng_pattern, cn_name in behavior_eng:
|
|
|
|
|
+ if f'行为_{cn_name}' not in result:
|
|
|
|
|
+ eng_match = re.search(eng_pattern, text)
|
|
|
|
|
+ if eng_match:
|
|
|
|
|
+ result[f'行为_{cn_name}'] = eng_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 家庭环境 (10维度) =====
|
|
|
|
|
+ # 格式: "亲密9分 情感表达7分 和谐5分..."
|
|
|
|
|
+ family_dims = ['亲密', '情感表达', '和谐', '独立性', '成就向导',
|
|
|
|
|
+ '文化氛围', '娱乐活动', '道德观念', '家务安排', '家庭规则']
|
|
|
|
|
+ # 方法1: 逐个匹配 "维度名+数字+分"
|
|
|
|
|
+ for dim in family_dims:
|
|
|
|
|
+ f_match = re.search(rf'{re.escape(dim)}\s*(\d+)\s*分', text)
|
|
|
|
|
+ if f_match:
|
|
|
|
|
+ result[f'家庭_{dim}'] = f_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 方法2: 在家庭环境相关页面批量提取数字序列
|
|
|
|
|
+ if len([k for k in result if k.startswith('家庭_')]) < 5:
|
|
|
|
|
+ if page_texts and len(page_texts) > 5:
|
|
|
|
|
+ page6_text = page_texts[5]
|
|
|
|
|
+ family_nums = re.findall(r'(\d+)\s*分', page6_text)
|
|
|
|
|
+ valid = [n for n in family_nums if 1 <= int(n) <= 15]
|
|
|
|
|
+ if len(valid) >= 10:
|
|
|
|
|
+ for i, dim in enumerate(family_dims):
|
|
|
|
|
+ if f'家庭_{dim}' not in result and i < len(valid):
|
|
|
|
|
+ result[f'家庭_{dim}'] = valid[i]
|
|
|
|
|
+
|
|
|
|
|
+ return result
|
|
|
|
|
+
|
|
|
|
|
+def extract_b3_data(text, page_texts=None):
|
|
|
|
|
+ """提取B3报告数据 - 核心学习能力(执行功能+学习动机+学习策略)"""
|
|
|
|
|
+ result = {}
|
|
|
|
|
+ result.update(extract_score_and_percentile(text))
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 执行功能 (3维度, 百分位) =====
|
|
|
|
|
+ # 格式: "抑制控制 94% 工作记忆 89% 认知灵活性 60%"
|
|
|
|
|
+ exec_dims = ['抑制控制', '工作记忆', '认知灵活性']
|
|
|
|
|
+ for dim in exec_dims:
|
|
|
|
|
+ ef_match = re.search(rf'{re.escape(dim)}\s*(\d+)%', text)
|
|
|
|
|
+ if ef_match:
|
|
|
|
|
+ result[f'{dim}_pct'] = ef_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 学习动机 (3维度, 十分制) =====
|
|
|
|
|
+ # 格式: "深层动机\n我的得分:8分" 或 "深层动机 8分"
|
|
|
|
|
+ motivation_dims = ['深层动机', '表面动机', '自我效能感']
|
|
|
|
|
+ for dim in motivation_dims:
|
|
|
|
|
+ # 方法1: "深层动机...我的得分:8分"
|
|
|
|
|
+ lm_match = re.search(
|
|
|
|
|
+ rf'{re.escape(dim)}.*?我的得分[::]\s*(\d+(?:\.\d+)?)',
|
|
|
|
|
+ text,
|
|
|
|
|
+ re.DOTALL
|
|
|
|
|
+ )
|
|
|
|
|
+ if lm_match:
|
|
|
|
|
+ result[dim] = lm_match.group(1)
|
|
|
|
|
+ else:
|
|
|
|
|
+ # 方法2: "深层动机 8分"
|
|
|
|
|
+ lm_match2 = re.search(rf'{re.escape(dim)}\s*(\d+(?:\.\d+)?)\s*分', text)
|
|
|
|
|
+ if lm_match2:
|
|
|
|
|
+ result[dim] = lm_match2.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 学习策略 (3维度, 十分制) =====
|
|
|
|
|
+ # 格式: "深层方法与策略...我的得分:6.8分"
|
|
|
|
|
+ strategy_dims = ['深层方法与策略', '表面方法与策略', '学习自我调节']
|
|
|
|
|
+ for dim in strategy_dims:
|
|
|
|
|
+ ls_match = re.search(
|
|
|
|
|
+ rf'{re.escape(dim)}.*?我的得分[::]\s*(\d+(?:\.\d+)?)',
|
|
|
|
|
+ text,
|
|
|
|
|
+ re.DOTALL
|
|
|
|
|
+ )
|
|
|
|
|
+ if ls_match:
|
|
|
|
|
+ result[dim] = ls_match.group(1)
|
|
|
|
|
+ else:
|
|
|
|
|
+ ls_match2 = re.search(rf'{re.escape(dim)}\s*(\d+(?:\.\d+)?)\s*分', text)
|
|
|
|
|
+ if ls_match2:
|
|
|
|
|
+ result[dim] = ls_match2.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ return result
|
|
|
|
|
+
|
|
|
|
|
+# ── 成长型思维仪表盘连续分数(基于表盘指针位置)──
|
|
|
|
|
+# 方法:从Layout B B4报告第12页提取1090×320半圆形仪表盘图像,
|
|
|
|
|
+# 测量三角指针的指向角度,通过atan2映射到0-100连续分数。
|
|
|
|
|
+# 校准基准:陈嘉梁(成长型倾向)指针角度≈-6.68°→score=44.0
|
|
|
|
|
+
|
|
|
|
|
+
|
|
|
|
|
+def _extract_1090x320_image(doc):
|
|
|
|
|
+ """从PDF文档中提取1090×320仪表盘图像的RGB像素数据。
|
|
|
|
|
+
|
|
|
|
|
+ Returns:
|
|
|
|
|
+ PIL.Image对象 (RGB模式) 或 None
|
|
|
|
|
+ """
|
|
|
|
|
+ try:
|
|
|
|
|
+ page = doc[12]
|
|
|
|
|
+ for img_info in page.get_images():
|
|
|
|
|
+ xref = img_info[0]
|
|
|
|
|
+ data = doc.extract_image(xref)
|
|
|
|
|
+ w, h = data['width'], data['height']
|
|
|
|
|
+ if (w, h) == (1090, 320):
|
|
|
|
|
+ img = Image.open(io.BytesIO(data['image']))
|
|
|
|
|
+ if img.mode != 'RGB':
|
|
|
|
|
+ img = img.convert('RGB')
|
|
|
|
|
+ return img
|
|
|
|
|
+ except Exception:
|
|
|
|
|
+ pass
|
|
|
|
|
+ return None
|
|
|
|
|
+
|
|
|
|
|
+
|
|
|
|
|
+def _measure_needle_position(img, center_x=544):
|
|
|
|
|
+ """从1090×320仪表盘图像中检测三角指针位置。
|
|
|
|
|
+
|
|
|
|
|
+ 原理:指针在图像中表现为左右刻度弧线之间的第三段彩色区域。
|
|
|
|
|
+ 从上往下扫描(y=60-220),找到指针与弧线分离最清晰的y坐标,
|
|
|
|
|
+ 返回指针横截面中心位置。
|
|
|
|
|
+
|
|
|
|
|
+ Returns:
|
|
|
|
|
+ (needle_x, needle_y, needle_width) 或 (None, None, None)
|
|
|
|
|
+ """
|
|
|
|
|
+ W, H = img.size
|
|
|
|
|
+
|
|
|
|
|
+ for y in range(60, 220):
|
|
|
|
|
+ runs = []
|
|
|
|
|
+ in_run = False
|
|
|
|
|
+ run_start = 0
|
|
|
|
|
+ x_min = max(0, center_x - 250)
|
|
|
|
|
+ x_max = min(W, center_x + 250)
|
|
|
|
|
+ for x in range(x_min, x_max):
|
|
|
|
|
+ p = img.getpixel((x, y))
|
|
|
|
|
+ if p[0] > 30 or p[1] > 30 or p[2] > 30:
|
|
|
|
|
+ if not in_run:
|
|
|
|
|
+ run_start = x
|
|
|
|
|
+ in_run = True
|
|
|
|
|
+ else:
|
|
|
|
|
+ if in_run:
|
|
|
|
|
+ runs.append((run_start, x - 1))
|
|
|
|
|
+ in_run = False
|
|
|
|
|
+ if in_run:
|
|
|
|
|
+ runs.append((run_start, x_max - 1))
|
|
|
|
|
+
|
|
|
|
|
+ big = [(s, e) for s, e in runs if e - s + 1 > 5]
|
|
|
|
|
+
|
|
|
|
|
+ if len(big) == 3:
|
|
|
|
|
+ left_end = big[0][1]
|
|
|
|
|
+ needle = big[1]
|
|
|
|
|
+ right_start = big[2][0]
|
|
|
|
|
+ if left_end < needle[0] and needle[1] < right_start:
|
|
|
|
|
+ needle_mid = (needle[0] + needle[1]) / 2.0
|
|
|
|
|
+ needle_width = needle[1] - needle[0] + 1
|
|
|
|
|
+ return needle_mid, y, needle_width
|
|
|
|
|
+
|
|
|
|
|
+ return None, None, None
|
|
|
|
|
+
|
|
|
|
|
+
|
|
|
|
|
+def compute_growth_mindset_score_from_needle(doc):
|
|
|
|
|
+ """从PDF文档的仪表盘图像计算成长型思维连续分数。
|
|
|
|
|
+
|
|
|
|
|
+ 三步流程:
|
|
|
|
|
+ 1. 提取1090×320仪表盘图像
|
|
|
|
|
+ 2. 检测三角指针位置(3段分离法)
|
|
|
|
|
+ 3. 计算指针角度并映射到0-100连续分数
|
|
|
|
|
+
|
|
|
|
|
+ 校准基准:
|
|
|
|
|
+ - 陈嘉梁(成长型倾向44分): 指针中心偏移center_x=-18.5px, y=97
|
|
|
|
|
+ → atan2(-18.5, 193.8-97) = atan2(-18.5, 96.8) = -10.8°
|
|
|
|
|
+ → score = (-10.8 + 90) / 180 * 100 = 44.0
|
|
|
|
|
+
|
|
|
|
|
+ Returns:
|
|
|
|
|
+ float 分数(0-100) 或 None(无法检测)
|
|
|
|
|
+ """
|
|
|
|
|
+ img = _extract_1090x320_image(doc)
|
|
|
|
|
+ if img is None:
|
|
|
|
|
+ return None
|
|
|
|
|
+
|
|
|
|
|
+ # 固定参数:表盘中心x坐标和指针旋转中心y坐标(从陈嘉梁标定)
|
|
|
|
|
+ center_x = 544
|
|
|
|
|
+ center_y = 193.8 # 指针旋转中心y坐标(由陈嘉梁44分标定)
|
|
|
|
|
+
|
|
|
|
|
+ needle_x, needle_y, _ = _measure_needle_position(img, center_x)
|
|
|
|
|
+ if needle_x is None:
|
|
|
|
|
+ return None
|
|
|
|
|
+
|
|
|
|
|
+ dx = needle_x - center_x
|
|
|
|
|
+ dy = center_y - needle_y
|
|
|
|
|
+ if dy <= 0:
|
|
|
|
|
+ return None
|
|
|
|
|
+
|
|
|
|
|
+ angle_rad = math.atan2(dx, dy)
|
|
|
|
|
+ angle_deg = math.degrees(angle_rad)
|
|
|
|
|
+
|
|
|
|
|
+ # 映射到0-100: 角度-90°(左) → 0分, 0°(上) → 50分, 90°(右) → 100分
|
|
|
|
|
+ score = (angle_deg + 90) / 180.0 * 100.0
|
|
|
|
|
+ return max(0.0, min(100.0, round(score, 1)))
|
|
|
|
|
+
|
|
|
|
|
+
|
|
|
|
|
+def extract_growth_mindset_from_b4_vec(pdf_path):
|
|
|
|
|
+ """分析B4 PDF矢量图形提取成长型思维倾向(Layout B第12页的复选框+条形图颜色)
|
|
|
|
|
+
|
|
|
|
|
+ 研究发现: Layout B (16页) 的113份文件中:
|
|
|
|
|
+ - 59份: 第12页为纯说明性内容, 无矢量复选框 → 无成长型思维数据
|
|
|
|
|
+ - 43份: 复选框在LEFT(固定型)teal高亮, RIGHT(成长型)黄色 → 固定型倾向
|
|
|
|
|
+ - 11份: 复选框在LEFT黄色, RIGHT teal高亮 → 成长型倾向
|
|
|
|
|
+
|
|
|
|
|
+ 参数:
|
|
|
|
|
+ pdf_path: PDF文件路径
|
|
|
|
|
+
|
|
|
|
|
+ 返回: '固定型倾向' / '成长型倾向' / '' (无数据)
|
|
|
|
|
+ """
|
|
|
|
|
+ try:
|
|
|
|
|
+ doc = fitz.open(pdf_path)
|
|
|
|
|
+ # 仅Layout B (16页) 可能有生长型思维矢量图形
|
|
|
|
|
+ if len(doc) != 16:
|
|
|
|
|
+ doc.close()
|
|
|
|
|
+ return ''
|
|
|
|
|
+
|
|
|
|
|
+ page = doc[12] # 成长型思维页面 (Layout B)
|
|
|
|
|
+ paths = page.get_drawings()
|
|
|
|
|
+
|
|
|
|
|
+ # 1) 检测teal色条形图宽度 (3种: 103/135/151)
|
|
|
|
|
+ bar_width = 0
|
|
|
|
|
+ for path in paths:
|
|
|
|
|
+ fill = path.get('fill')
|
|
|
|
|
+ rect = path.get('rect')
|
|
|
|
|
+ if not fill:
|
|
|
|
|
+ continue
|
|
|
|
|
+ w = rect[2] - rect[0]
|
|
|
|
|
+ if w > 80 and fill[0] > 0.1 and fill[0] < 0.3 and fill[1] > 0.6:
|
|
|
|
|
+ bar_width = round(w)
|
|
|
|
|
+ break
|
|
|
|
|
+
|
|
|
|
|
+ if bar_width == 0 or bar_width == 135:
|
|
|
|
|
+ # 135 = 没有复选框的Layout变体 → 无个性化数据
|
|
|
|
|
+ doc.close()
|
|
|
|
|
+ return ''
|
|
|
|
|
+
|
|
|
|
|
+ # 2) 检测复选框颜色 (48x17的矩形)
|
|
|
|
|
+ # LEFT侧 (x~79) = 固定型思维特征; RIGHT侧 (x~467) = 成长型思维特征
|
|
|
|
|
+ teal_on_left = 0
|
|
|
|
|
+ teal_on_right = 0
|
|
|
|
|
+ for path in paths:
|
|
|
|
|
+ fill = path.get('fill')
|
|
|
|
|
+ rect = path.get('rect')
|
|
|
|
|
+ if not fill:
|
|
|
|
|
+ continue
|
|
|
|
|
+ w = round(rect[2] - rect[0])
|
|
|
|
|
+ h = round(rect[3] - rect[1])
|
|
|
|
|
+ if w == 48 and h == 17:
|
|
|
|
|
+ if rect[0] < 200: # LEFT = 固定型
|
|
|
|
|
+ if fill[0] > 0.1 and fill[0] < 0.3 and fill[1] > 0.6:
|
|
|
|
|
+ teal_on_left += 1
|
|
|
|
|
+ elif rect[0] > 300: # RIGHT = 成长型
|
|
|
|
|
+ if fill[0] > 0.1 and fill[0] < 0.3 and fill[1] > 0.6:
|
|
|
|
|
+ teal_on_right += 1
|
|
|
|
|
+
|
|
|
|
|
+ # 判断: 合并条形图宽度和复选框证据
|
|
|
|
|
+ result = ''
|
|
|
|
|
+ if teal_on_left >= 3 and teal_on_right == 0:
|
|
|
|
|
+ result = '固定型倾向'
|
|
|
|
|
+ elif teal_on_right >= 3 and teal_on_left == 0:
|
|
|
|
|
+ result = '成长型倾向'
|
|
|
|
|
+ elif bar_width == 151:
|
|
|
|
|
+ result = '固定型倾向'
|
|
|
|
|
+ elif bar_width == 103:
|
|
|
|
|
+ result = '成长型倾向'
|
|
|
|
|
+
|
|
|
|
|
+ doc.close()
|
|
|
|
|
+ return result
|
|
|
|
|
+ except Exception as e:
|
|
|
|
|
+ print(f"Error extracting growth mindset from {pdf_path}: {e}")
|
|
|
|
|
+ return ''
|
|
|
|
|
+
|
|
|
|
|
+
|
|
|
|
|
+def extract_b4_data(text, page_texts=None, pdf_path=None):
|
|
|
|
|
+ """提取B4报告数据 - 核心认知能力+自我概念+自驱力+成长型思维"""
|
|
|
|
|
+ result = {}
|
|
|
|
|
+ result.update(extract_score_and_percentile(text))
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 核心认知维度 (同A1, 6维度) =====
|
|
|
|
|
+ cognitive_dims = ['感知觉', '注意力', '记忆力', '推理能力', '空间能力', '加工速度']
|
|
|
|
|
+
|
|
|
|
|
+ # 格式2 (优先): summary认知模型页面 "感知觉 | Perception\n...描述...\n百分位(%)\n14"
|
|
|
|
|
+ # 该页面百分位数据最准确, 不会被detail页面的刻度标签(0%-100%)干扰
|
|
|
|
|
+ for dim in cognitive_dims:
|
|
|
|
|
+ pct_match = re.search(
|
|
|
|
|
+ rf'{re.escape(dim)}\s*\|\s*\w+[\s\S]*?百分位(%)\s*\n\s*(\d+)',
|
|
|
|
|
+ text
|
|
|
|
|
+ )
|
|
|
|
|
+ if pct_match:
|
|
|
|
|
+ result[f'{dim}_pct'] = pct_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 格式1 (后补): detail页面, 仅当格式2未提取到时使用
|
|
|
|
|
+ # 格式1a: "我的感知觉得分\n17|18%" 或 带乱码分隔符 "17ح18%"
|
|
|
|
|
+ # 格式1b: "我的加工速度得分\n33 | 33 | 88%" (原始分|标准分|百分位)
|
|
|
|
|
+ # 使用两步法避免回溯bug: 先截取"得分"后文本段, 再分别搜score和pct
|
|
|
|
|
+ # 注意: 11页B4只有推理能力/空间能力/加工速度3维有detail页, 所以只补这3维
|
|
|
|
|
+ for dim in cognitive_dims:
|
|
|
|
|
+ detail_section = re.search(
|
|
|
|
|
+ rf'我的{re.escape(dim)}得分(.{{1,300}})',
|
|
|
|
|
+ text, re.DOTALL
|
|
|
|
|
+ )
|
|
|
|
|
+ if detail_section:
|
|
|
|
|
+ section_text = detail_section.group(1)
|
|
|
|
|
+ # 提取第一个数字%作为百分位 (首个出现的是实际分数)
|
|
|
|
|
+ pct_match = re.search(r'(\d+)%', section_text)
|
|
|
|
|
+ if pct_match:
|
|
|
|
|
+ if f'{dim}_pct' not in result:
|
|
|
|
|
+ result[f'{dim}_pct'] = pct_match.group(1)
|
|
|
|
|
+ # 提取第一个数字作为原始分
|
|
|
|
|
+ score_match = re.search(r'(\d+)', section_text)
|
|
|
|
|
+ if score_match:
|
|
|
|
|
+ result[f'{dim}_score'] = score_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 自我概念 (6维度, 0-10分制, 第11页柱状图) =====
|
|
|
|
|
+ # B4第11页(索引10)是横向柱状图, 每维度一个横条, 0-10刻度
|
|
|
|
|
+ # 分数印制在横条右端上方, font_size≈9.6(区别于坐标轴标签6.2和维度名称8.6)
|
|
|
|
|
+ # 使用positioned text提取, 禁止使用%匹配(会拿到认知页的百分位值)
|
|
|
|
|
+ self_concept_out = ['行为表现', '能力与学校', '躯体外貌', '情绪状态', '合群', '幸福与满足']
|
|
|
|
|
+ if pdf_path:
|
|
|
|
|
+ try:
|
|
|
|
|
+ doc = fitz.open(pdf_path)
|
|
|
|
|
+ n_pages = len(doc)
|
|
|
|
|
+ # 两种B4格式:
|
|
|
|
|
+ # 16页格式: self-concept在第11页(索引10), 分数font_size≈9.6
|
|
|
|
|
+ # 11页格式: self-concept在第6页(索引5), 分数font_size≈6.4
|
|
|
|
|
+ candidates = [(10, 8, 11), (5, 6, 8)] if n_pages > 12 else [(5, 6, 8), (10, 8, 11)]
|
|
|
|
|
+ found = False
|
|
|
|
|
+ for page_idx, fs_min, fs_max in candidates:
|
|
|
|
|
+ if page_idx >= n_pages:
|
|
|
|
|
+ continue
|
|
|
|
|
+ page = doc[page_idx]
|
|
|
|
|
+ blocks = page.get_text('dict')['blocks']
|
|
|
|
|
+ scores = []
|
|
|
|
|
+ for b in blocks:
|
|
|
|
|
+ if b['type'] == 0:
|
|
|
|
|
+ for line in b['lines']:
|
|
|
|
|
+ for span in line['spans']:
|
|
|
|
|
+ span_text = span['text'].strip()
|
|
|
|
|
+ if not span_text:
|
|
|
|
|
+ continue
|
|
|
|
|
+ size = span['size']
|
|
|
|
|
+ x = span['bbox'][0]
|
|
|
|
|
+ if (fs_min <= size <= fs_max and x > 250
|
|
|
|
|
+ and span_text.isdigit() and 0 <= int(span_text) <= 10):
|
|
|
|
|
+ scores.append((span['bbox'][1], int(span_text)))
|
|
|
|
|
+ scores.sort(key=lambda s: s[0])
|
|
|
|
|
+ if len(scores) >= 6:
|
|
|
|
|
+ for i, dim in enumerate(self_concept_out):
|
|
|
|
|
+ result[f'自我概念_{dim}'] = str(scores[i][1])
|
|
|
|
|
+ found = True
|
|
|
|
|
+ break
|
|
|
|
|
+ if not found:
|
|
|
|
|
+ # 第三次尝试: 在所有页面中搜索"自我概念"字样, 提取相邻数字
|
|
|
|
|
+ for pi in range(n_pages):
|
|
|
|
|
+ page_text = doc[pi].get_text()
|
|
|
|
|
+ if '自我概念' in page_text and '行为表现' in page_text:
|
|
|
|
|
+ blocks = doc[pi].get_text('dict')['blocks']
|
|
|
|
|
+ scores = []
|
|
|
|
|
+ for b in blocks:
|
|
|
|
|
+ if b['type'] == 0:
|
|
|
|
|
+ for line in b['lines']:
|
|
|
|
|
+ for span in line['spans']:
|
|
|
|
|
+ st = span['text'].strip()
|
|
|
|
|
+ if not st:
|
|
|
|
|
+ continue
|
|
|
|
|
+ size = span['size']
|
|
|
|
|
+ x = span['bbox'][0]
|
|
|
|
|
+ if (5 <= size <= 12 and x > 250
|
|
|
|
|
+ and st.isdigit() and 0 <= int(st) <= 10):
|
|
|
|
|
+ scores.append((span['bbox'][1], int(st)))
|
|
|
|
|
+ scores.sort(key=lambda s: s[0])
|
|
|
|
|
+ if len(scores) >= 6:
|
|
|
|
|
+ for i, dim in enumerate(self_concept_out):
|
|
|
|
|
+ result[f'自我概念_{dim}'] = str(scores[i][1])
|
|
|
|
|
+ break
|
|
|
|
|
+ doc.close()
|
|
|
|
|
+ except Exception as e:
|
|
|
|
|
+ print(f" [WARN] B4 self-concept extraction failed: {e}")
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 自驱力 =====
|
|
|
|
|
+ all_scores = re.findall(r'我的得分[::](\d+\.?\d*)', text)
|
|
|
|
|
+ if len(all_scores) >= 3:
|
|
|
|
|
+ result['自主性'] = all_scores[0]
|
|
|
|
|
+ result['胜任感'] = all_scores[1]
|
|
|
|
|
+ result['归属感'] = all_scores[2]
|
|
|
|
|
+ elif len(all_scores) >= 1:
|
|
|
|
|
+ result['自主性'] = all_scores[0]
|
|
|
|
|
+
|
|
|
|
|
+ if '自主性' not in result:
|
|
|
|
|
+ parts = text.split('Autonomy')
|
|
|
|
|
+ if len(parts) >= 3:
|
|
|
|
|
+ match = re.search(r'(\d+\.?\d*)', parts[2])
|
|
|
|
|
+ if match:
|
|
|
|
|
+ result['自主性'] = match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ if '胜任感' not in result:
|
|
|
|
|
+ parts = text.split('Competence')
|
|
|
|
|
+ if len(parts) >= 3:
|
|
|
|
|
+ match = re.search(r'(\d+\.?\d*)', parts[3] if len(parts) > 3 else parts[2])
|
|
|
|
|
+ if match:
|
|
|
|
|
+ result['胜任感'] = match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ if '归属感' not in result:
|
|
|
|
|
+ parts = text.split('Relatedness')
|
|
|
|
|
+ if len(parts) >= 3:
|
|
|
|
|
+ match = re.search(r'(\d+\.?\d*)', parts[3] if len(parts) > 3 else parts[2])
|
|
|
|
|
+ if match:
|
|
|
|
|
+ result['归属感'] = match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 成长型思维(连续分数) =====
|
|
|
|
|
+ # 从1090×320仪表盘图像检测指针位置,映射到0-100刻度
|
|
|
|
|
+ # Layout A (11页) 的B4文件无仪表盘图像 → 留空
|
|
|
|
|
+ result['成长型思维'] = ''
|
|
|
|
|
+
|
|
|
|
|
+ if pdf_path:
|
|
|
|
|
+ try:
|
|
|
|
|
+ doc = fitz.open(pdf_path)
|
|
|
|
|
+ score = compute_growth_mindset_score_from_needle(doc)
|
|
|
|
|
+ doc.close()
|
|
|
|
|
+ if score is not None:
|
|
|
|
|
+ result['成长型思维'] = score
|
|
|
|
|
+ except Exception:
|
|
|
|
|
+ pass
|
|
|
|
|
+
|
|
|
|
|
+ return result
|
|
|
|
|
+
|
|
|
|
|
+def extract_b6_data(text, page_texts=None):
|
|
|
|
|
+ """提取B6报告数据 - 职业发展(兴趣+能力+价值观)"""
|
|
|
|
|
+ result = {}
|
|
|
|
|
+ result.update(extract_score_and_percentile(text))
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 职业兴趣 (Holland 6类型) =====
|
|
|
|
|
+ # 格式: "NO.1: 艺术型 Artistic NO.2: 社会型 Social 7分 5分"
|
|
|
|
|
+ interest_dims = {
|
|
|
|
|
+ '艺术型': '兴趣_艺术型',
|
|
|
|
|
+ '社会型': '兴趣_社会型',
|
|
|
|
|
+ '事业型': '兴趣_事业型',
|
|
|
|
|
+ '常规型': '兴趣_常规型',
|
|
|
|
|
+ '现实型': '兴趣_现实型',
|
|
|
|
|
+ '研究型': '兴趣_研究型',
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ # 方法1: NO.X模式
|
|
|
|
|
+ for cn_name, field_name in interest_dims.items():
|
|
|
|
|
+ # Match "X型 ... X分" where X分 follows the interest type name
|
|
|
|
|
+ h_match = re.search(
|
|
|
|
|
+ rf'{re.escape(cn_name)}\s+\w+\s+(\d+)\s*分',
|
|
|
|
|
+ text
|
|
|
|
|
+ )
|
|
|
|
|
+ if h_match:
|
|
|
|
|
+ result[field_name] = h_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 多元智能 (8能力) =====
|
|
|
|
|
+ # 格式: "内省能力 INTRAPERSONAL 8分 人际关系能力 INTERPERSONAL 4分"
|
|
|
|
|
+ ability_dims = {
|
|
|
|
|
+ '内省能力': '能力_内省',
|
|
|
|
|
+ '空间能力': '能力_空间',
|
|
|
|
|
+ '音乐能力': '能力_音乐',
|
|
|
|
|
+ '人际关系能力': '能力_人际关系',
|
|
|
|
|
+ '自然能力': '能力_自然',
|
|
|
|
|
+ '身体运动能力': '能力_身体运动',
|
|
|
|
|
+ '语言能力': '能力_语言',
|
|
|
|
|
+ '逻辑数学能力': '能力_逻辑数学',
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ for cn_name, field_name in ability_dims.items():
|
|
|
|
|
+ ab_match = re.search(
|
|
|
|
|
+ rf'{re.escape(cn_name)}\s+\w+\s+(\d+)\s*分',
|
|
|
|
|
+ text
|
|
|
|
|
+ )
|
|
|
|
|
+ if ab_match:
|
|
|
|
|
+ result[field_name] = ab_match.group(1)
|
|
|
|
|
+ else:
|
|
|
|
|
+ # 更简单: "内省能力 8分"
|
|
|
|
|
+ ab_match2 = re.search(
|
|
|
|
|
+ rf'{re.escape(cn_name)}\s*(\d+)\s*分',
|
|
|
|
|
+ text
|
|
|
|
|
+ )
|
|
|
|
|
+ if ab_match2:
|
|
|
|
|
+ # Make sure it's not matching a multi-digit number in a different context
|
|
|
|
|
+ score = ab_match2.group(1)
|
|
|
|
|
+ if 1 <= int(score) <= 15:
|
|
|
|
|
+ result[field_name] = score
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 职业价值观 (可选) =====
|
|
|
|
|
+ # 格式: 在"我的职业价值观"区域有因子得分如 "8.11\n声望地位" "5.41\n美的追求"
|
|
|
|
|
+ # 取第一个因子得分作为代表性分数(限1-15分)
|
|
|
|
|
+ if '我的职业价值观' in text:
|
|
|
|
|
+ m = re.search(r'我的职业价值观.*?(\d+(?:\.\d+)?)', text, re.DOTALL)
|
|
|
|
|
+ if m:
|
|
|
|
|
+ val = float(m.group(1))
|
|
|
|
|
+ if 1 <= val <= 15:
|
|
|
|
|
+ result['职业价值观'] = m.group(1)
|
|
|
|
|
+ # 后备: "最高分"附近的明确数值
|
|
|
|
|
+ if '职业价值观' not in result:
|
|
|
|
|
+ m = re.search(r'职业价值观.*?最高分[^\d]*?(\d+(?:\.\d+)?)', text, re.DOTALL)
|
|
|
|
|
+ if m:
|
|
|
|
|
+ val = float(m.group(1))
|
|
|
|
|
+ if 1 <= val <= 15:
|
|
|
|
|
+ result['职业价值观'] = m.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ return result
|
|
|
|
|
+
|
|
|
|
|
+def extract_c1_data(text, page_texts=None):
|
|
|
|
|
+ """提取C1报告数据 - 校园版综合(认知+人格+自驱力+自我概念)"""
|
|
|
|
|
+ result = {}
|
|
|
|
|
+ result.update(extract_score_and_percentile(text))
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 核心认知维度 (同A1, 6维度) =====
|
|
|
|
|
+ cognitive_dims = ['感知觉', '注意力', '记忆力', '推理能力', '空间能力', '加工速度']
|
|
|
|
|
+
|
|
|
|
|
+ # 格式1: "我的感知觉得分 41 | 14%"
|
|
|
|
|
+ for dim in cognitive_dims:
|
|
|
|
|
+ detail_match = re.search(
|
|
|
|
|
+ rf'我的{re.escape(dim)}得分\s*(\d+)\s*\|\s*(\d+)%',
|
|
|
|
|
+ text
|
|
|
|
|
+ )
|
|
|
|
|
+ if detail_match:
|
|
|
|
|
+ result[f'{dim}_score'] = detail_match.group(1)
|
|
|
|
|
+ result[f'{dim}_pct'] = detail_match.group(2)
|
|
|
|
|
+
|
|
|
|
|
+ # 格式2: summary页面 "感知觉 | Perception\n百分位(%)\n14"
|
|
|
|
|
+ for dim in cognitive_dims:
|
|
|
|
|
+ if f'{dim}_pct' not in result:
|
|
|
|
|
+ pct_match = re.search(
|
|
|
|
|
+ rf'{re.escape(dim)}\s*\|\s*\w+\s*\n\s*百分位(%)\s*\n\s*(\d+)',
|
|
|
|
|
+ text
|
|
|
|
|
+ )
|
|
|
|
|
+ if pct_match:
|
|
|
|
|
+ result[f'{dim}_pct'] = pct_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 大五人格 (同A2) =====
|
|
|
|
|
+ big5_names = ['开放性', '宜人性', '责任心', '外倾性', '神经质']
|
|
|
|
|
+ if page_texts:
|
|
|
|
|
+ for pt in page_texts:
|
|
|
|
|
+ for dim in big5_names:
|
|
|
|
|
+ if dim not in result:
|
|
|
|
|
+ b5_match = re.search(rf'您在[""「]{re.escape(dim)}[""」].*?得分是\s*(\d+(?:\.\d+)?)\s*分', pt)
|
|
|
|
|
+ if b5_match:
|
|
|
|
|
+ result[dim] = b5_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 自驱力 (同B4) =====
|
|
|
|
|
+ all_scores = re.findall(r'我的得分[::](\d+\.?\d*)', text)
|
|
|
|
|
+ if len(all_scores) >= 3:
|
|
|
|
|
+ result['自主性'] = all_scores[0]
|
|
|
|
|
+ result['胜任感'] = all_scores[1]
|
|
|
|
|
+ result['归属感'] = all_scores[2]
|
|
|
|
|
+ elif len(all_scores) >= 1:
|
|
|
|
|
+ result['自主性'] = all_scores[0]
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 自我概念 (同B4) =====
|
|
|
|
|
+ self_concept_start = text.find('SELF-CONCEPT')
|
|
|
|
|
+ if self_concept_start >= 0:
|
|
|
|
|
+ before_section = text[:self_concept_start]
|
|
|
|
|
+ pct_matches = re.findall(r'(\d+)%', before_section)
|
|
|
|
|
+
|
|
|
|
|
+ sc_dims = ['行为表现', '能力与学校', '躯体外貌', '情绪状态', '合群', '幸福与满足']
|
|
|
|
|
+ if len(pct_matches) >= 12:
|
|
|
|
|
+ high_pcts = pct_matches[-12:]
|
|
|
|
|
+ for i, dim in enumerate(sc_dims):
|
|
|
|
|
+ idx = i * 2 + 1
|
|
|
|
|
+ if idx < len(high_pcts):
|
|
|
|
|
+ result[f'自我概念_{dim}'] = high_pcts[idx]
|
|
|
|
|
+ elif len(pct_matches) >= 6:
|
|
|
|
|
+ for i, dim in enumerate(sc_dims):
|
|
|
|
|
+ if i < len(pct_matches):
|
|
|
|
|
+ result[f'自我概念_{dim}'] = pct_matches[-(6-i)]
|
|
|
|
|
+
|
|
|
|
|
+ return result
|
|
|
|
|
+
|
|
|
|
|
+# ===== B5 人际关系指南针图子维度提取 =====
|
|
|
|
|
+# 校准: 每个象限的 fill_avg → score 线性映射系数
|
|
|
|
|
+# v2: 扫描起点r=150跳过中心色块; 因子由extract_b5_data传入per-student值
|
|
|
|
|
+_B5_QUAD_FACTORS = {
|
|
|
|
|
+ '母亲关系': 0.006974,
|
|
|
|
|
+ '父亲关系': 0.005834,
|
|
|
|
|
+ '师生关系': 0.006820,
|
|
|
|
|
+ '同伴关系': 0.005049,
|
|
|
|
|
+}
|
|
|
|
|
+_B5_SUB_DIMS = {
|
|
|
|
|
+ '母亲关系': ['控制压迫', '情感疏离', '价值冲突', '双重标准', '过度期待'],
|
|
|
|
|
+ '父亲关系': ['控制压迫', '情感疏离', '价值冲突', '双重标准', '过度期待'],
|
|
|
|
|
+ '师生关系': ['回避型', '敌对型', '高控制型', '关系疏离型'],
|
|
|
|
|
+ '同伴关系': ['回避型', '被排斥型', '攻击型', '边缘型', '特殊因素型'],
|
|
|
|
|
+}
|
|
|
|
|
+# 象限角度范围 (图像坐标系: 0°=右, 90°=下, 180°=左, 270°=上)
|
|
|
|
|
+_B5_QUAD_ANGLES = {
|
|
|
|
|
+ '母亲关系': (185, 265),
|
|
|
|
|
+ '父亲关系': (140, 178),
|
|
|
|
|
+ '师生关系': (275, 350),
|
|
|
|
|
+ '同伴关系': (2, 42),
|
|
|
|
|
+}
|
|
|
|
|
+_B5_CENTER = (857, 758)
|
|
|
|
|
+
|
|
|
|
|
+def _measure_fill_at_angle(img, cx, cy, angle_deg):
|
|
|
|
|
+ """沿指定角度测量填充范围(像素半径)。
|
|
|
|
|
+ 从r=150开始扫描以跳过中心色块(避免环线阻断),
|
|
|
|
|
+ 25px间隙容忍用于跨越填充分区内的环线。"""
|
|
|
|
|
+ angle = math.radians(angle_deg)
|
|
|
|
|
+ w, h = img.size
|
|
|
|
|
+ max_r = int(min(w-cx, cx, cy, h-cy)) - 2
|
|
|
|
|
+ fill_end = None
|
|
|
|
|
+ gap = 0
|
|
|
|
|
+ for r in range(150, max_r, 1):
|
|
|
|
|
+ x = int(cx + r * math.cos(angle))
|
|
|
|
|
+ y = int(cy + r * math.sin(angle))
|
|
|
|
|
+ if not (0 <= x < w and 0 <= y < h):
|
|
|
|
|
+ break
|
|
|
|
|
+ px = img.getpixel((x, y))
|
|
|
|
|
+ is_white = all(c > 230 for c in px[:3])
|
|
|
|
|
+ is_gray = all(155 < c < 195 for c in px[:3])
|
|
|
|
|
+ is_black = all(c < 20 for c in px[:3])
|
|
|
|
|
+ is_colored = not (is_white or is_gray or is_black)
|
|
|
|
|
+ if is_colored:
|
|
|
|
|
+ fill_end = r
|
|
|
|
|
+ gap = 0
|
|
|
|
|
+ elif fill_end is not None:
|
|
|
|
|
+ gap += 1
|
|
|
|
|
+ if gap > 25:
|
|
|
|
|
+ break
|
|
|
|
|
+ return fill_end
|
|
|
|
|
+
|
|
|
|
|
+def _extract_b5_compass_scores(doc, quad_factors=None):
|
|
|
|
|
+ """从B5 PDF文档提取人际关系指南针19个子维度分数。
|
|
|
|
|
+ 返回 { '人际_母亲关系_控制压迫': score, ... }"""
|
|
|
|
|
+ result = {}
|
|
|
|
|
+ page = doc[8] # 第9页
|
|
|
|
|
+ imgs = page.get_images()
|
|
|
|
|
+ if len(imgs) < 4:
|
|
|
|
|
+ return result
|
|
|
|
|
+ base = doc.extract_image(imgs[3][0])
|
|
|
|
|
+ img = Image.open(io.BytesIO(base['image']))
|
|
|
|
|
+ cx, cy = _B5_CENTER
|
|
|
|
|
+
|
|
|
|
|
+ # 从页面文本提取per-student factor(如果未提供)
|
|
|
|
|
+ if quad_factors is None:
|
|
|
|
|
+ text = page.get_text()
|
|
|
|
|
+ lines = text.split('\n')
|
|
|
|
|
+ text_scores = {}
|
|
|
|
|
+ for rel in ['母亲关系', '父亲关系', '师生关系', '同伴关系']:
|
|
|
|
|
+ for i, line in enumerate(lines):
|
|
|
|
|
+ if line.strip() == rel:
|
|
|
|
|
+ for j in range(i+1, min(i+5, len(lines))):
|
|
|
|
|
+ try:
|
|
|
|
|
+ text_scores[rel] = float(lines[j].strip())
|
|
|
|
|
+ break
|
|
|
|
|
+ except ValueError:
|
|
|
|
|
+ pass
|
|
|
|
|
+ break
|
|
|
|
|
+ # 计算quad平均fill和per-student factor
|
|
|
|
|
+ quad_factors = {}
|
|
|
|
|
+ for qname, (a1, a2) in _B5_QUAD_ANGLES.items():
|
|
|
|
|
+ fills = []
|
|
|
|
|
+ for a in range(a1, a2+1):
|
|
|
|
|
+ fe = _measure_fill_at_angle(img, cx, cy, a)
|
|
|
|
|
+ if fe:
|
|
|
|
|
+ fills.append(fe)
|
|
|
|
|
+ avg = sum(fills)/len(fills) if fills else 1
|
|
|
|
|
+ txt = text_scores.get(qname, 3.0)
|
|
|
|
|
+ quad_factors[qname] = txt / avg if avg > 0 else _B5_QUAD_FACTORS[qname]
|
|
|
|
|
+
|
|
|
|
|
+ for quad_name, (a_start, a_end) in _B5_QUAD_ANGLES.items():
|
|
|
|
|
+ dims = _B5_SUB_DIMS[quad_name]
|
|
|
|
|
+ n = len(dims)
|
|
|
|
|
+ step = (a_end - a_start) / (n + 1)
|
|
|
|
|
+ factor = quad_factors.get(quad_name, _B5_QUAD_FACTORS[quad_name])
|
|
|
|
|
+
|
|
|
|
|
+ for i, dim_name in enumerate(dims):
|
|
|
|
|
+ angle = a_start + (i + 1) * step
|
|
|
|
|
+ fills = []
|
|
|
|
|
+ for offset in [-2, 0, 2]:
|
|
|
|
|
+ fe = _measure_fill_at_angle(img, cx, cy, angle + offset)
|
|
|
|
|
+ if fe is not None:
|
|
|
|
|
+ fills.append(fe)
|
|
|
|
|
+ avg_fill = sum(fills) / len(fills) if fills else 0
|
|
|
|
|
+ score = min(5.0, max(1.0, round(avg_fill * factor, 1)))
|
|
|
|
|
+ result[f'人际_{quad_name}_{dim_name}'] = score
|
|
|
|
|
+ return result
|
|
|
|
|
+
|
|
|
|
|
+def extract_b5_data(text, page_texts=None, pdf_path=None):
|
|
|
|
|
+ """提取B5报告数据 - 青春期挑战(情绪调节+学业压力+人际关系+社交+睡眠+运动+网络依赖)"""
|
|
|
|
|
+ result = {}
|
|
|
|
|
+ result.update(extract_score_and_percentile(text))
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 情绪调节策略 (2维度) =====
|
|
|
|
|
+ er_match = re.search(r'认知重评.*?Cognitive Reappraisal\s*(\d+(?:\.\d+)?)\s*分', text, re.DOTALL)
|
|
|
|
|
+ if not er_match:
|
|
|
|
|
+ er_match = re.search(r'认知重评\s*(\d+(?:\.\d+)?)\s*分', text)
|
|
|
|
|
+ if er_match:
|
|
|
|
|
+ result['认知重评'] = er_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ es_match = re.search(r'表达抑制.*?Expressive Suppression\s*(\d+(?:\.\d+)?)\s*分', text, re.DOTALL)
|
|
|
|
|
+ if not es_match:
|
|
|
|
|
+ es_match = re.search(r'表达抑制\s*(\d+(?:\.\d+)?)\s*分', text)
|
|
|
|
|
+ if es_match:
|
|
|
|
|
+ result['表达抑制'] = es_match.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 学业压力源 (5维度) =====
|
|
|
|
|
+ stress_section = re.search(r'学业压力源[\s\S]*?(?=人际关系问题)', text)
|
|
|
|
|
+ if not stress_section:
|
|
|
|
|
+ stress_section = re.search(r'学业压力[\s\S]*?(?=社交能力|睡眠)', text)
|
|
|
|
|
+ stress_text = stress_section.group() if stress_section else text
|
|
|
|
|
+ stress_dims = ['学业负担', '家庭期望', '师生关系', '自我期望', '同伴竞争']
|
|
|
|
|
+ for dim in stress_dims:
|
|
|
|
|
+ m = re.search(re.escape(dim) + r'\s*\n\s*(\d+(?:\.\d+)?)', stress_text)
|
|
|
|
|
+ if m:
|
|
|
|
|
+ result[f'学业压力_{dim}'] = m.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 人际关系问题 (4维度) =====
|
|
|
|
|
+ rel_section = re.search(r'人际关系问题[\s\S]*?(?=社交能力)', text)
|
|
|
|
|
+ rel_text = rel_section.group() if rel_section else text
|
|
|
|
|
+ relation_dims = ['母亲关系', '父亲关系', '师生关系', '同伴关系']
|
|
|
|
|
+ # 数据区格式: "维度名\n描述句。\n分数" 需跳过intro段中的同名词
|
|
|
|
|
+ for dim in relation_dims:
|
|
|
|
|
+ m = re.search(re.escape(dim) + r'\n[^\n]*。\n(\d+(?:\.\d+)?)', rel_text)
|
|
|
|
|
+ if m:
|
|
|
|
|
+ result[f'人际_{dim}'] = m.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 社交能力 (5维度) =====
|
|
|
|
|
+ soc_section = re.search(r'社交能力[\s\S]*?(?=睡眠)', text)
|
|
|
|
|
+ soc_text = soc_section.group() if soc_section else text
|
|
|
|
|
+ soc_m = re.search(r'同龄人平均水平\s*\n\s*(\d+(?:\.\d+)?)\s*\n\s*(\d+(?:\.\d+)?)\s*\n\s*(\d+(?:\.\d+)?)\s*\n\s*(\d+(?:\.\d+)?)\s*\n\s*(\d+(?:\.\d+)?)', soc_text)
|
|
|
|
|
+ if soc_m:
|
|
|
|
|
+ soc_dims = ['主动交往', '情感支持', '情感表达', '表达影响', '冲突解决']
|
|
|
|
|
+ for i, dim in enumerate(soc_dims):
|
|
|
|
|
+ result[f'社交_{dim}'] = soc_m.group(i+1)
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 睡眠数据 =====
|
|
|
|
|
+ sleep_hours = re.search(r'睡眠时长[^。]*?(\d+(?:\.\d+)?)\s*小时', text)
|
|
|
|
|
+ if sleep_hours:
|
|
|
|
|
+ result['睡眠_时长'] = sleep_hours.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ sleep_eff = re.search(r'睡眠效率[^。]*?(\d+(?:\.\d+)?)%', text)
|
|
|
|
|
+ if sleep_eff:
|
|
|
|
|
+ result['睡眠_效率'] = sleep_eff.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ sleep_lat = re.search(r'入睡时间[^。]*?(\d+)-(\d+)分钟', text)
|
|
|
|
|
+ if sleep_lat:
|
|
|
|
|
+ result['入睡时间_min'] = sleep_lat.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 睡眠定性维度:取维度名后第三行(值行)
|
|
|
|
|
+ sleep_qual = {
|
|
|
|
|
+ '主观睡眠质量': '睡眠_主观质量',
|
|
|
|
|
+ '日间功能障碍': '睡眠_日间功能',
|
|
|
|
|
+ '催眠药物': '睡眠_催眠药物',
|
|
|
|
|
+ }
|
|
|
|
|
+ for kw, col in sleep_qual.items():
|
|
|
|
|
+ m = re.search(re.escape(kw) + r'\n[^\n]*\n[^\n]*\n([^\n]+)', text)
|
|
|
|
|
+ if m:
|
|
|
|
|
+ result[col] = m.group(1).strip()
|
|
|
|
|
+
|
|
|
|
|
+ sleep_disorder = re.search(r'睡眠障碍[^。]*?(\S+)', text)
|
|
|
|
|
+ if sleep_disorder:
|
|
|
|
|
+ result['睡眠_障碍'] = sleep_disorder.group(1).strip()
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 运动数据 =====
|
|
|
|
|
+ ex_names = ['久坐', '步行', '中等强度', '高强度']
|
|
|
|
|
+ for ex_name in ex_names:
|
|
|
|
|
+ m = re.search(re.escape(ex_name) + r'\n(\d+)\n天/周\n(.+)', text)
|
|
|
|
|
+ if m:
|
|
|
|
|
+ result[f'运动_{ex_name}_频率'] = m.group(1)
|
|
|
|
|
+ result[f'运动_{ex_name}_时长'] = m.group(2).strip()
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 网络依赖 =====
|
|
|
|
|
+ internet_total = re.search(r'(\d+)分\s*\n\s*强迫使用', text)
|
|
|
|
|
+ if internet_total:
|
|
|
|
|
+ result['网络依赖_总分'] = internet_total.group(1)
|
|
|
|
|
+
|
|
|
|
|
+ # 用上方的分数匹配子维度(分数在维度名之上)
|
|
|
|
|
+ # 格式: "44分\n强迫使用\n...43分\n时间管理问题\n...35分\n戒断反应"
|
|
|
|
|
+ internet_sub = re.findall(r'(\d+)分\s*\n\s*(强迫使用|时间管理问题|戒断反应)', text)
|
|
|
|
|
+ for score, dim_name in internet_sub:
|
|
|
|
|
+ col_map = {'强迫使用': '网络依赖_强迫使用', '时间管理问题': '网络依赖_时间管理', '戒断反应': '网络依赖_戒断反应'}
|
|
|
|
|
+ if dim_name in col_map:
|
|
|
|
|
+ result[col_map[dim_name]] = score
|
|
|
|
|
+
|
|
|
|
|
+ # ===== 人际关系指南针19个子维度(第9页图像) =====
|
|
|
|
|
+ if pdf_path and os.path.exists(pdf_path):
|
|
|
|
|
+ try:
|
|
|
|
|
+ doc = fitz.open(pdf_path)
|
|
|
|
|
+ if len(doc) > 8:
|
|
|
|
|
+ compass_scores = _extract_b5_compass_scores(doc)
|
|
|
|
|
+ result.update(compass_scores)
|
|
|
|
|
+ doc.close()
|
|
|
|
|
+ except Exception:
|
|
|
|
|
+ pass # 图像提取失败不影响已有数据
|
|
|
|
|
+
|
|
|
|
|
+ return result
|
|
|
|
|
+
|
|
|
|
|
+def determine_report_type(filename):
|
|
|
|
|
+ """根据文件名判断报告类型(后备方案,内容指纹优先)"""
|
|
|
|
|
+ # 已被 type_detector.detect_type() 替代
|
|
|
|
|
+ # 保留作为纯后备(当文本层损坏导致 detect_type 回退时调用 fallback_by_filename)
|
|
|
|
|
+ from type_detector import fallback_by_filename
|
|
|
|
|
+ return fallback_by_filename(filename)
|
|
|
|
|
+
|
|
|
|
|
+def extract_date_from_filename(filename):
|
|
|
|
|
+ """从文件名中提取测评日期 YYYYMMDD"""
|
|
|
|
|
+ # 匹配 8位连续数字日期 (如 20250320)
|
|
|
|
|
+ m = re.search(r'(\d{8})', filename)
|
|
|
|
|
+ if m:
|
|
|
|
|
+ return m.group(1)
|
|
|
|
|
+ # 匹配 YYYY-MM-DD 或 YYYY_MM_DD 格式(文件名中常见带连字符)
|
|
|
|
|
+ m = re.search(r'(\d{4})[-_](\d{2})[-_](\d{2})', filename)
|
|
|
|
|
+ if m:
|
|
|
|
|
+ return m.group(1) + m.group(2) + m.group(3)
|
|
|
|
|
+ return ''
|
|
|
|
|
+
|
|
|
|
|
+def extract_all_data(pdf_path, filename):
|
|
|
|
|
+ """提取所有类型报告的数据"""
|
|
|
|
|
+ text = extract_text_from_pdf(pdf_path)
|
|
|
|
|
+ page_texts = extract_text_from_pdf(pdf_path, return_pages=True)
|
|
|
|
|
+
|
|
|
|
|
+ name = extract_name_from_filename(filename)
|
|
|
|
|
+ birthday = extract_birthday_from_pdf(text, page_texts)
|
|
|
|
|
+ report_type, _ = detect_type(text, filename)
|
|
|
|
|
+
|
|
|
|
|
+ # 基础信息
|
|
|
|
|
+ row = {
|
|
|
|
|
+ 'filename': filename,
|
|
|
|
|
+ '姓名': name,
|
|
|
|
|
+ '生日': birthday,
|
|
|
|
|
+ '报告类型': report_type,
|
|
|
|
|
+ '测评日期': extract_date_from_filename(filename),
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ # 根据类型提取数据
|
|
|
|
|
+ if report_type == 'A1':
|
|
|
|
|
+ data = extract_a1_data(text, page_texts)
|
|
|
|
|
+ elif report_type == 'A2':
|
|
|
|
|
+ data = extract_a2_data(text, page_texts)
|
|
|
|
|
+ elif report_type == 'B2':
|
|
|
|
|
+ data = extract_b2_data(text, page_texts)
|
|
|
|
|
+ elif report_type == 'B3':
|
|
|
|
|
+ data = extract_b3_data(text, page_texts)
|
|
|
|
|
+ elif report_type == 'B4':
|
|
|
|
|
+ data = extract_b4_data(text, page_texts, pdf_path)
|
|
|
|
|
+ elif report_type == 'B5':
|
|
|
|
|
+ data = extract_b5_data(text, page_texts, pdf_path)
|
|
|
|
|
+ elif report_type == 'B6':
|
|
|
|
|
+ data = extract_b6_data(text, page_texts)
|
|
|
|
|
+ elif report_type == 'C1':
|
|
|
|
|
+ data = extract_c1_data(text, page_texts)
|
|
|
|
|
+ else:
|
|
|
|
|
+ data = {}
|
|
|
|
|
+
|
|
|
|
|
+ row.update(data)
|
|
|
|
|
+ return row
|
|
|
|
|
+
|
|
|
|
|
+def main():
|
|
|
|
|
+ # 自动检测项目根目录 (脚本在 scripts/ 下)
|
|
|
|
|
+ script_dir = os.path.dirname(os.path.abspath(__file__))
|
|
|
|
|
+ project_dir = os.path.dirname(os.path.dirname(script_dir)) # 项目根目录(上溯两级:scripts/reports/ → scripts/ → 根)
|
|
|
|
|
+ reports_dir = os.path.join(project_dir, "报告")
|
|
|
|
|
+ output_file = os.path.join(project_dir, "素材库", "测评数据_优化提取.csv")
|
|
|
|
|
+
|
|
|
|
|
+ # 获取所有PDF文件
|
|
|
|
|
+ pdf_files = []
|
|
|
|
|
+ for f in os.listdir(reports_dir):
|
|
|
|
|
+ if f.endswith('.pdf') and '案例' not in f and '示例' not in f and 'eStatement' not in f:
|
|
|
|
|
+ pdf_files.append((os.path.join(reports_dir, f), f))
|
|
|
|
|
+
|
|
|
|
|
+ print(f"Found {len(pdf_files)} PDF files")
|
|
|
|
|
+
|
|
|
|
|
+ all_results = []
|
|
|
|
|
+ type_counts = {}
|
|
|
|
|
+
|
|
|
|
|
+ for i, (pdf_path, filename) in enumerate(pdf_files):
|
|
|
|
|
+ if i % 100 == 0:
|
|
|
|
|
+ print(f"Processing {i}/{len(pdf_files)}...")
|
|
|
|
|
+
|
|
|
|
|
+ try:
|
|
|
|
|
+ row = extract_all_data(pdf_path, filename)
|
|
|
|
|
+ rt = row.get('报告类型', 'Unknown')
|
|
|
|
|
+ type_counts[rt] = type_counts.get(rt, 0) + 1
|
|
|
|
|
+ all_results.append(row)
|
|
|
|
|
+ except Exception as e:
|
|
|
|
|
+ print(f"Error processing {filename}: {e}")
|
|
|
|
|
+
|
|
|
|
|
+ print(f"\n=== Report Type Counts ===")
|
|
|
|
|
+ for rt, count in sorted(type_counts.items()):
|
|
|
|
|
+ print(f"{rt}: {count}")
|
|
|
|
|
+
|
|
|
|
|
+ # 写入CSV
|
|
|
|
|
+ if all_results:
|
|
|
|
|
+ all_keys = set()
|
|
|
|
|
+ for row in all_results:
|
|
|
|
|
+ all_keys.update(row.keys())
|
|
|
|
|
+
|
|
|
|
|
+ columns = ['filename', '姓名', '生日', '报告类型', '测评日期', '总分', '百分位']
|
|
|
|
|
+ for k in sorted(all_keys):
|
|
|
|
|
+ if k not in columns:
|
|
|
|
|
+ columns.append(k)
|
|
|
|
|
+
|
|
|
|
|
+ with open(output_file, 'w', encoding='utf-8-sig', newline='') as f:
|
|
|
|
|
+ writer = csv.DictWriter(f, fieldnames=columns)
|
|
|
|
|
+ writer.writeheader()
|
|
|
|
|
+ writer.writerows(all_results)
|
|
|
|
|
+
|
|
|
|
|
+ print(f"\nWritten {len(all_results)} records to {output_file}")
|
|
|
|
|
+
|
|
|
|
|
+if __name__ == "__main__":
|
|
|
|
|
+ main()
|