Browse Source

feat: DAN报告上传 - 创建DanReportParseService (A2/B4解析)

支持心维度(A2)大五人格/社会关系/情绪状态/综合能力和智维度(B4)自我概念/成长型思维/自驱力的PDF解析

Ultraworked with Sisyphus

Co-authored-by: Sisyphus <clio-agent@sisyphuslabs.ai>
User 2 months ago
parent
commit
e3a0bcf1b7

+ 589 - 0
cfc-backend/src/main/java/com/etotem/cfc/service/DanReportParseService.java

@@ -0,0 +1,589 @@
+package com.etotem.cfc.service;
+
+import lombok.extern.slf4j.Slf4j;
+import org.apache.pdfbox.pdmodel.PDDocument;
+import org.apache.pdfbox.text.PDFTextStripper;
+import org.springframework.stereotype.Service;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.util.*;
+import java.util.regex.Matcher;
+import java.util.regex.Pattern;
+
+/**
+ * DAN 测评报告 PDF 解析服务
+ * 支持:
+ * - A2报告(心/mind):大五人格、社会关系、情绪状态、综合能力
+ * - B4报告(智/wisdom):自我概念、成长型思维、自驱力
+ *
+ * 解析结果以结构化数据项列表返回,灵活存储于 structured_analysis JSON 字段。
+ */
+@Slf4j
+@Service
+public class DanReportParseService {
+
+    /**
+     * 解析 PDF 输入流,返回结构化数据
+     */
+    public DanParsedReport parse(InputStream inputStream, String dimension) throws IOException {
+        try (PDDocument document = PDDocument.load(inputStream)) {
+            PDFTextStripper stripper = new PDFTextStripper();
+            stripper.setSortByPosition(true);
+            String fullText = stripper.getText(document);
+            return parseText(fullText, dimension);
+        }
+    }
+
+    /**
+     * 从纯文本解析 DAN 报告
+     */
+    public DanParsedReport parseText(String fullText, String dimension) {
+        String[] lines = fullText.split("\\r?\\n");
+        List<String> lineList = new ArrayList<>();
+        for (String line : lines) {
+            String trimmed = line.trim();
+            if (!trimmed.isEmpty()) {
+                // 去除PDF提取中的特殊Unicode字符
+                trimmed = trimmed.replaceAll("[\\uF000-\\uFFFF]", "").trim();
+                if (!trimmed.isEmpty()) {
+                    lineList.add(trimmed);
+                }
+            }
+        }
+
+        if ("mind".equals(dimension)) {
+            return parseA2Report(lineList, fullText);
+        } else if ("wisdom".equals(dimension)) {
+            return parseB4Report(lineList, fullText);
+        }
+        return DanParsedReport.empty();
+    }
+
+    // ======================== A2 报告解析(心/mind) ========================
+
+    private DanParsedReport parseA2Report(List<String> lines, String fullText) {
+        List<DataItem> items = new ArrayList<>();
+        Map<String, Object> extra = new LinkedHashMap<>();
+        StringBuilder summary = new StringBuilder();
+        StringBuilder suggestions = new StringBuilder();
+
+        // 1. 基本信息和日期
+        String reportDate = findFieldValue(lines, "测评日期", "报告日期", "评估日期");
+        String name = findFieldValue(lines, "姓名", "学生姓名", "被评估人");
+        String gender = findFieldValue(lines, "性别");
+        String age = findFieldValue(lines, "年龄");
+
+        extra.put("name", name);
+        extra.put("gender", gender);
+        extra.put("age", age);
+        extra.put("reportDate", reportDate);
+
+        // 2. 大五人格 (Big Five)
+        // 查找大五人格区域
+        int bigFiveStart = findSectionStart(lines, "大五人格", "人格特质", "人格分析");
+        if (bigFiveStart >= 0) {
+            Map<String, String> bigFive = parseBigFive(lines, bigFiveStart);
+            for (Map.Entry<String, String> entry : bigFive.entrySet()) {
+                items.add(new DataItem(entry.getKey(), entry.getKey(), entry.getValue(), "bigFive"));
+            }
+            extra.put("bigFive", bigFive);
+        } else {
+            // 全局搜索大五维度
+            Map<String, String> bigFive = searchBigFiveGlobally(lines);
+            if (!bigFive.isEmpty()) {
+                for (Map.Entry<String, String> entry : bigFive.entrySet()) {
+                    items.add(new DataItem(entry.getKey(), entry.getKey(), entry.getValue(), "bigFive"));
+                }
+                extra.put("bigFive", bigFive);
+            }
+        }
+
+        // 3. 社会关系(与父母/同伴)
+        int socialStart = findSectionStart(lines, "社会关系", "人际关系", "家庭关系", "同伴关系");
+        if (socialStart >= 0) {
+            Map<String, String> social = parseSocialRelations(lines, socialStart);
+            if (!social.isEmpty()) {
+                for (Map.Entry<String, String> entry : social.entrySet()) {
+                    items.add(new DataItem(entry.getKey(), entry.getKey(), entry.getValue(), "social"));
+                }
+                extra.put("social", social);
+            }
+        }
+
+        // 4. 情绪状态
+        int emotionStart = findSectionStart(lines, "情绪状态", "情绪", "情感");
+        if (emotionStart >= 0) {
+            Map<String, String> emotion = parseEmotionState(lines, emotionStart);
+            if (!emotion.isEmpty()) {
+                for (Map.Entry<String, String> entry : emotion.entrySet()) {
+                    items.add(new DataItem(entry.getKey(), entry.getKey(), entry.getValue(), "emotion"));
+                }
+                extra.put("emotion", emotion);
+            }
+        }
+
+        // 5. 综合能力
+        int abilityStart = findSectionStart(lines, "综合能力", "综合评估", "综合");
+        if (abilityStart >= 0) {
+            Map<String, String> ability = parseComprehensiveAbility(lines, abilityStart);
+            if (!ability.isEmpty()) {
+                for (Map.Entry<String, String> entry : ability.entrySet()) {
+                    items.add(new DataItem(entry.getKey(), entry.getKey(), entry.getValue(), "ability"));
+                }
+                extra.put("ability", ability);
+            }
+        }
+
+        // 6. 通用正则:提取所有 "维度名: 分数" 模式
+        Pattern scorePattern = Pattern.compile("([\\u4e00-\\u9fa5]{2,8})[::]\\s*(\\d+(\\.\\d+)?)");
+        Matcher matcher = scorePattern.matcher(fullText);
+        while (matcher.find()) {
+            String key = matcher.group(1).trim();
+            String val = matcher.group(2).trim();
+            // 去重:避免与已解析项重复
+            boolean exists = false;
+            for (DataItem item : items) {
+                if (item.getName().equals(key)) {
+                    exists = true;
+                    break;
+                }
+            }
+            if (!exists && !key.contains("日期") && !key.contains("姓名")) {
+                items.add(new DataItem("score_" + key.hashCode(), key, val, "auto"));
+            }
+        }
+
+        // 总结和建议
+        summary.append(extractSection(fullText, "测评总结", "成长建议"));
+        suggestions.append(extractSection(fullText, "成长建议", null));
+
+        return new DanParsedReport("A2", "mind", items, extra, summary.toString(), suggestions.toString());
+    }
+
+    // ======================== B4 报告解析(智/wisdom) ========================
+
+    private DanParsedReport parseB4Report(List<String> lines, String fullText) {
+        List<DataItem> items = new ArrayList<>();
+        Map<String, Object> extra = new LinkedHashMap<>();
+        StringBuilder summary = new StringBuilder();
+        StringBuilder suggestions = new StringBuilder();
+
+        // 1. 基本信息和日期
+        String reportDate = findFieldValue(lines, "测评日期", "报告日期", "评估日期");
+        String name = findFieldValue(lines, "姓名", "学生姓名", "被评估人");
+        extra.put("name", name);
+        extra.put("reportDate", reportDate);
+
+        // 2. 自我概念 (Self-Concept)
+        int selfConceptStart = findSectionStart(lines, "自我概念", "自我认知", "自我意识");
+        if (selfConceptStart >= 0) {
+            Map<String, String> selfConcept = parseSelfConcept(lines, selfConceptStart);
+            if (!selfConcept.isEmpty()) {
+                for (Map.Entry<String, String> entry : selfConcept.entrySet()) {
+                    items.add(new DataItem(entry.getKey(), entry.getKey(), entry.getValue(), "selfConcept"));
+                }
+                extra.put("selfConcept", selfConcept);
+            }
+        }
+
+        // 3. 成长型思维 (Growth Mindset)
+        int mindsetStart = findSectionStart(lines, "成长思维", "成长型思维", "思维模式");
+        if (mindsetStart >= 0) {
+            Map<String, String> mindset = parseGrowthMindset(lines, mindsetStart);
+            if (!mindset.isEmpty()) {
+                for (Map.Entry<String, String> entry : mindset.entrySet()) {
+                    items.add(new DataItem(entry.getKey(), entry.getKey(), entry.getValue(), "growthMindset"));
+                }
+                extra.put("growthMindset", mindset);
+            }
+        }
+
+        // 4. 自驱力 (Self-Driving)
+        int drivingStart = findSectionStart(lines, "自驱力", "自主性", "内驱力", "自我驱动");
+        if (drivingStart >= 0) {
+            Map<String, String> selfDriving = parseSelfDriving(lines, drivingStart);
+            if (!selfDriving.isEmpty()) {
+                for (Map.Entry<String, String> entry : selfDriving.entrySet()) {
+                    items.add(new DataItem(entry.getKey(), entry.getKey(), entry.getValue(), "selfDriving"));
+                }
+                extra.put("selfDriving", selfDriving);
+            }
+        }
+
+        // 5. 通用正则提取
+        Pattern scorePattern = Pattern.compile("([\\u4e00-\\u9fa5]{2,8})[::]\\s*(\\d+(\\.\\d+)?)");
+        Matcher matcher = scorePattern.matcher(fullText);
+        while (matcher.find()) {
+            String key = matcher.group(1).trim();
+            String val = matcher.group(2).trim();
+            boolean exists = false;
+            for (DataItem item : items) {
+                if (item.getName().equals(key)) {
+                    exists = true;
+                    break;
+                }
+            }
+            if (!exists && !key.contains("日期") && !key.contains("姓名")) {
+                items.add(new DataItem("score_" + key.hashCode(), key, val, "auto"));
+            }
+        }
+
+        summary.append(extractSection(fullText, "测评总结", "成长建议"));
+        suggestions.append(extractSection(fullText, "成长建议", null));
+
+        return new DanParsedReport("B4", "wisdom", items, extra, summary.toString(), suggestions.toString());
+    }
+
+    // ======================== 解析辅助方法 ========================
+
+    /**
+     * 在大五人格区域内解析维度:开放/尽责/外倾/宜人/神经质
+     */
+    private Map<String, String> parseBigFive(List<String> lines, int startIdx) {
+        Map<String, String> result = new LinkedHashMap<>();
+        String[] labels = {"开放性", "尽责性", "外倾性", "宜人性", "神经质"};
+        Set<String> labelSet = new HashSet<>(Arrays.asList(labels));
+
+        for (int i = startIdx; i < Math.min(startIdx + 30, lines.size()); i++) {
+            String line = lines.get(i);
+            for (String label : labels) {
+                if (line.contains(label)) {
+                    String score = extractScore(line);
+                    if (score != null) {
+                        result.put(label, score);
+                    }
+                    break;
+                }
+            }
+        }
+        return result;
+    }
+
+    /**
+     * 全局搜索大五维度(当找不到明确的大五区域时)
+     */
+    private Map<String, String> searchBigFiveGlobally(List<String> lines) {
+        Map<String, String> result = new LinkedHashMap<>();
+        String[] labels = {"开放性", "尽责性", "外倾性", "宜人性", "神经质",
+                "开放", "尽责", "外倾", "宜人", "神经"};
+        for (String line : lines) {
+            for (String label : labels) {
+                if (line.contains(label)) {
+                    String score = extractScore(line);
+                    if (score != null) {
+                        // 使用更精确的维度名
+                        String displayName = label;
+                        if (label.equals("开放")) displayName = "开放性";
+                        else if (label.equals("尽责")) displayName = "尽责性";
+                        else if (label.equals("外倾")) displayName = "外倾性";
+                        else if (label.equals("宜人")) displayName = "宜人性";
+                        else if (label.equals("神经")) displayName = "神经质";
+                        result.put(displayName, score);
+                    }
+                }
+            }
+        }
+        return result;
+    }
+
+    /**
+     * 解析社会关系:与父母信任、沟通、亲近等
+     */
+    private Map<String, String> parseSocialRelations(List<String> lines, int startIdx) {
+        Map<String, String> result = new LinkedHashMap<>();
+        String[] keywords = {"母亲", "父亲", "同伴", "信任", "沟通", "亲近", "疏远"};
+        for (int i = startIdx; i < Math.min(startIdx + 40, lines.size()); i++) {
+            String line = lines.get(i);
+            for (String kw : keywords) {
+                if (line.contains(kw)) {
+                    String score = extractScore(line);
+                    if (score != null) {
+                        // 提取更精确的键名
+                        String key = line.replaceAll("\\d+", "").replaceAll("[::()()]", "").trim();
+                        if (key.length() > 2 && key.length() < 20) {
+                            result.put(key, score);
+                        } else {
+                            result.put(kw, score);
+                        }
+                    }
+                    break;
+                }
+            }
+        }
+        return result;
+    }
+
+    /**
+     * 解析情绪状态
+     */
+    private Map<String, String> parseEmotionState(List<String> lines, int startIdx) {
+        Map<String, String> result = new LinkedHashMap<>();
+        String[] keywords = {"自卑", "自信", "抑郁", "愉快", "焦虑", "安详", "无力感", "掌控感"};
+        for (int i = startIdx; i < Math.min(startIdx + 30, lines.size()); i++) {
+            String line = lines.get(i);
+            for (String kw : keywords) {
+                if (line.contains(kw)) {
+                    String score = extractScore(line);
+                    if (score != null) {
+                        result.put(kw, score);
+                    }
+                    break;
+                }
+            }
+        }
+        return result;
+    }
+
+    /**
+     * 解析综合能力
+     */
+    private Map<String, String> parseComprehensiveAbility(List<String> lines, int startIdx) {
+        Map<String, String> result = new LinkedHashMap<>();
+        String[] keywords = {"情绪调节", "抗挫折", "内驱力", "社会适应", "同理心", "自律"};
+        for (int i = startIdx; i < Math.min(startIdx + 30, lines.size()); i++) {
+            String line = lines.get(i);
+            for (String kw : keywords) {
+                if (line.contains(kw)) {
+                    String score = extractScore(line);
+                    if (score != null) {
+                        result.put(kw, score);
+                    }
+                    break;
+                }
+            }
+        }
+        return result;
+    }
+
+    /**
+     * 解析自我概念(B4)
+     */
+    private Map<String, String> parseSelfConcept(List<String> lines, int startIdx) {
+        Map<String, String> result = new LinkedHashMap<>();
+        // 自我概念通常包含6个维度
+        for (int i = startIdx; i < Math.min(startIdx + 30, lines.size()); i++) {
+            String line = lines.get(i);
+            if (line.length() > 1 && line.length() < 15 && !line.matches(".*[\\d]{2,}.*")) {
+                String nextLine = i + 1 < lines.size() ? lines.get(i + 1) : "";
+                String score = extractScore(nextLine.isEmpty() ? line : nextLine);
+                if (score != null) {
+                    result.put(line, score);
+                    i++;
+                }
+            } else {
+                String score = extractScore(line);
+                if (score != null) {
+                    String name = line.replaceAll("\\d+", "").replaceAll("[::()().%]", "").trim();
+                    if (name.length() >= 2) {
+                        result.put(name, score);
+                    }
+                }
+            }
+        }
+        return result;
+    }
+
+    /**
+     * 解析成长型思维(B4)
+     */
+    private Map<String, String> parseGrowthMindset(List<String> lines, int startIdx) {
+        Map<String, String> result = new LinkedHashMap<>();
+        Pattern pattern = Pattern.compile("(成长[型思维]*|思维模式|智力[观看法]*|能力[观看法]*)[^\\d]*(\\d+)");
+        for (int i = startIdx; i < Math.min(startIdx + 20, lines.size()); i++) {
+            String line = lines.get(i);
+            Matcher m = pattern.matcher(line);
+            if (m.find()) {
+                result.put(m.group(1).trim(), m.group(2));
+            } else {
+                String score = extractScore(line);
+                if (score != null) {
+                    String name = line.replaceAll("\\d+\\.?\\d*", "").replaceAll("[::()()]", "").trim();
+                    if (name.length() >= 2 && name.length() < 20) {
+                        result.put(name, score);
+                    }
+                }
+            }
+        }
+        return result;
+    }
+
+    /**
+     * 解析自驱力(B4)
+     */
+    private Map<String, String> parseSelfDriving(List<String> lines, int startIdx) {
+        Map<String, String> result = new LinkedHashMap<>();
+        String[] keywords = {"自主性", "胜任感", "归属感", "自驱力", "自我驱动", "主动性"};
+        for (int i = startIdx; i < Math.min(startIdx + 30, lines.size()); i++) {
+            String line = lines.get(i);
+            for (String kw : keywords) {
+                if (line.contains(kw)) {
+                    String score = extractScore(line);
+                    if (score != null) {
+                        result.put(kw, score);
+                    }
+                    break;
+                }
+            }
+        }
+        return result;
+    }
+
+    // ======================== 通用工具方法 ========================
+
+    /**
+     * 从行中提取数字分数
+     */
+    private String extractScore(String line) {
+        if (line == null) return null;
+        // 先尝试匹配 "名称: 分数" 格式
+        Pattern p1 = Pattern.compile("[::]\\s*(\\d+(\\.\\d+)?)");
+        Matcher m1 = p1.matcher(line);
+        if (m1.find()) {
+            return m1.group(1);
+        }
+        // 再尝试单纯提取数字
+        Pattern p2 = Pattern.compile("(\\d+(\\.\\d+)?)");
+        Matcher m2 = p2.matcher(line);
+        if (m2.find()) {
+            String num = m2.group(1);
+            // 避免提取年份(4位数)或明显不是分数的数字
+            int val = Integer.parseInt(num.replace(".",""));
+            if (val >= 0 && val <= 100) {
+                return num;
+            }
+        }
+        return null;
+    }
+
+    /**
+     * 查找section标题所在行
+     */
+    private int findSectionStart(List<String> lines, String... headers) {
+        for (int i = 0; i < lines.size(); i++) {
+            String line = lines.get(i);
+            for (String header : headers) {
+                if (line.contains(header)) {
+                    return i + 1;
+                }
+            }
+        }
+        return -1;
+    }
+
+    /**
+     * 查找字段值(支持多个候选关键词)
+     */
+    private String findFieldValue(List<String> lines, String... keywords) {
+        for (String keyword : keywords) {
+            for (int i = 0; i < lines.size() - 1; i++) {
+                String line = lines.get(i);
+                if (line.contains(keyword + ":") || line.contains(keyword + ":")) {
+                    // 同行模式: "姓名: 张三"
+                    int colonIdx = Math.max(line.indexOf(':'), line.indexOf(':'));
+                    if (colonIdx >= 0 && colonIdx + 1 < line.length()) {
+                        String val = line.substring(colonIdx + 1).trim();
+                        if (!val.isEmpty()) return val;
+                    }
+                } else if (line.trim().equals(keyword) || line.trim().equals(keyword + ":") || line.trim().equals(keyword + ":")) {
+                    // 下一行模式
+                    String val = lines.get(i + 1).trim();
+                    if (!val.isEmpty() && val.length() < 50) return val;
+                }
+            }
+        }
+        return null;
+    }
+
+    /**
+     * 提取文本段落
+     */
+    private String extractSection(String text, String startMarker, String endMarker) {
+        int start = text.indexOf(startMarker);
+        if (start < 0) return "";
+        start += startMarker.length();
+        if (endMarker == null || endMarker.isEmpty()) {
+            return text.substring(start).replaceAll("^[\\s:\\n]+", "").trim();
+        }
+        int end = text.indexOf(endMarker, start);
+        if (end > start) {
+            return text.substring(start, end).replaceAll("^[\\s:\\n]+", "").trim();
+        }
+        return "";
+    }
+
+    // ======================== 内部数据类 ========================
+
+    /**
+     * 单项数据
+     */
+    public static class DataItem {
+        private String code;
+        private String name;
+        private String value;
+        private String category;
+
+        public DataItem() {}
+
+        public DataItem(String code, String name, String value, String category) {
+            this.code = code;
+            this.name = name;
+            this.value = value;
+            this.category = category;
+        }
+
+        public String getCode() { return code; }
+        public void setCode(String code) { this.code = code; }
+        public String getName() { return name; }
+        public void setName(String name) { this.name = name; }
+        public String getValue() { return value; }
+        public void setValue(String value) { this.value = value; }
+        public String getCategory() { return category; }
+        public void setCategory(String category) { this.category = category; }
+    }
+
+    /**
+     * 解析结果
+     */
+    public static class DanParsedReport {
+        private String reportType; // A2 / B4
+        private String dimension;  // mind / wisdom
+        private List<DataItem> items;
+        private Map<String, Object> extra;
+        private String summary;
+        private String suggestions;
+
+        public DanParsedReport() {}
+
+        public DanParsedReport(String reportType, String dimension, List<DataItem> items,
+                                Map<String, Object> extra, String summary, String suggestions) {
+            this.reportType = reportType;
+            this.dimension = dimension;
+            this.items = items;
+            this.extra = extra;
+            this.summary = summary;
+            this.suggestions = suggestions;
+        }
+
+        public static DanParsedReport empty() {
+            return new DanParsedReport("", "", new ArrayList<>(), new LinkedHashMap<>(), "", "");
+        }
+
+        public boolean isEmpty() {
+            return items == null || items.isEmpty();
+        }
+
+        public String getReportType() { return reportType; }
+        public void setReportType(String reportType) { this.reportType = reportType; }
+        public String getDimension() { return dimension; }
+        public void setDimension(String dimension) { this.dimension = dimension; }
+        public List<DataItem> getItems() { return items; }
+        public void setItems(List<DataItem> items) { this.items = items; }
+        public Map<String, Object> getExtra() { return extra; }
+        public void setExtra(Map<String, Object> extra) { this.extra = extra; }
+        public String getSummary() { return summary; }
+        public void setSummary(String summary) { this.summary = summary; }
+        public String getSuggestions() { return suggestions; }
+        public void setSuggestions(String suggestions) { this.suggestions = suggestions; }
+    }
+}