pdf_parser.py 32 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753
  1. """
  2. 菌群报告 PDF 解析器
  3. 从 extract_full_report_v5.py 提取核心逻辑,封装为可调用函数
  4. """
  5. import os
  6. import re
  7. from PyPDF2 import PdfReader
  8. # === 常量 ===
  9. # Kangxi 部首 → CJK 统一汉字(与 extract_full_report_v5.py 完全一致)
  10. RADICAL_MAP = {
  11. '\u2f18': '卜', '\u2f1f': '土', '\u2f24': '大', '\u2f26': '子',
  12. '\u2f29': '小', '\u2f2d': '山', '\u2f32': '干', '\u2f3c': '心',
  13. '\u2f42': '文', '\u2f46': '无', '\u2f4a': '木', '\u2f50': '比',
  14. '\u2f54': '水', '\u2f55': '火', '\u2f5c': '牛', '\u2f5f': '玉',
  15. '\u2f60': '瓜', '\u2f62': '甘', '\u2f63': '生', '\u2f64': '用',
  16. '\u2f69': '白', '\u2f6a': '皮', '\u2f6c': '目', '\u2f6f': '石',
  17. '\u2f75': '竹', '\u2f76': '米', '\u2f7a': '羊', '\u2f7b': '羽',
  18. '\u2f7c': '老', '\u2f7f': '耳', '\u2f81': '肉', '\u2f90': '衣',
  19. '\u2f95': '谷', '\u2f96': '豆', '\u2f9d': '身', '\u2fa6': '金',
  20. '\u2faf': '面', '\u2fb2': '韭', '\u2fb9': '香', '\u2fca': '黑',
  21. '\u2ec9': '贝', '\u2edd': '食', '\u2ee2': '马', '\u2ee5': '鱼',
  22. '\u2ee8': '麦', '\u2ee9': '黄', '\u2ef0': '龙',
  23. # 氏 radical U+2F52 → U+6C0F
  24. '\u2f52': '氏',
  25. }
  26. KNOWN_MACRO = ['碳水化合物', '蛋白质', '脂肪', '纤维素', '乳制品']
  27. KNOWN_AMINO = ['苏氨酸', '异亮氨酸', '亮氨酸', '赖氨酸', '蛋氨酸', '胱氨酸',
  28. '苯丙氨酸', '酪氨酸', '缬氨酸', '组氨酸', '丙氨酸', '丝氨酸', '甘氨酸',
  29. '脯氨酸', '谷氨酸', '天门冬氨酸', '天冬氨酸', '天冬酰胺', '谷氨酰胺',
  30. '精氨酸', '色氨酸']
  31. KNOWN_VITAMINS = ['维生素A', '维生素B1', '维生素B2', '维生素B5', '维生素B6',
  32. '叶酸', '维生素B12', '维生素C', '维生素D', '维生素K2', '维生素E']
  33. KNOWN_TRACE = ['铁', '锌']
  34. KNOWN_DISEASE_RISKS = ['炎症性肠炎', '肠易激综合征', '感染性腹泻', '自闭症',
  35. '抑郁症', '甲状腺疾病', '肺部感染或疾病', '自体免疫病', '结直肠癌',
  36. '肥胖', '便秘', '过敏', '失眠', '肝病', '肾病', '胃病', '胆病',
  37. '心脑血管疾病', 'II型糖尿病']
  38. KNOWN_BARRIER = ['肠道炎症水平', '肠道产气', '肠道屏障', '脂多糖LPS',
  39. '次级胆汁酸', '对甲酚(p-Cresol)', '吲哚', '苯酚', '腐胺', '硫化氢', '尸胺']
  40. KNOWN_SCFA = ['丁酸盐(Butyrate)', '丙酸盐(Propionate)', '乙酸盐(Acetate)', '异戊酸盐(Isovaleric)']
  41. KNOWN_NEURO = ['血清素(5-HT)', 'γ-氨基丁酸(GABA)', '谷氨酸(Glutamate)',
  42. '色氨酸(Tryptophan)', 'DOPAC', '多巴胺', '组胺(Histamine)', '一氧化氮',
  43. '喹啉(Quinolinic)', '维生素K2', '肌醇(Inositol)', '肾上腺素',
  44. '去甲肾上腺素', '乙酰胆碱', '皮质醇']
  45. KNOWN_ANTIBIOTICS = ['β-内酰胺酶类', '氨基糖苷类', '大环内酯类', '呋喃类',
  46. '喹诺酮类', '磺胺类', '甲氧苄啶类', '氯霉素类', '四环素类']
  47. KNOWN_PATHOGENS = ['幽门螺杆菌', '艰难梭菌', '沙门氏菌', '志贺氏菌', '弯曲杆菌']
  48. # === 菌群表标题(完整版 extract_full_report_v5.py 移植) ===
  49. BACTERIA_TABLE_TITLES = [
  50. '核心菌属构成表', '益生菌', '有害菌属构成表',
  51. '其它重要菌属构成表', '病原菌属构成表',
  52. ]
  53. PHYLUM_TABLE_TITLES = ['菌门构成表', '菌群门水平构成表', '门水平菌群构成']
  54. CLASS_TABLE_TITLES = ['菌纲构成表', '菌群纲水平构成表', '纲水平菌群构成']
  55. ORDER_TABLE_TITLES = ['菌目构成表', '菌群目水平构成表', '目水平菌群构成']
  56. FAMILY_TABLE_TITLES = ['菌科构成表', '菌群科水平构成表', '科水平菌群构成']
  57. GENUS_TABLE_TITLES = ['菌属构成表', '菌群属水平构成表', '属水平菌群构成']
  58. SPECIES_TABLE_TITLES = ['菌种构成表', '菌群种水平构成表', '种水平菌群构成']
  59. DISEASE_BACTERIA_TITLES = [
  60. '肥胖相关菌', '便秘相关菌', '抑郁相关菌', '过敏相关菌',
  61. '腹胀相关菌', '失眠相关菌', '肠道健康相关菌',
  62. '多动症相关菌', '自闭症相关菌',
  63. ]
  64. # === 食物推荐表(完整版 extract_full_report_v5.py 移植) ===
  65. COLUMNS_FOOD = ['名称', '分类', '推荐指数', '能量KJ', '蛋白g', '脂肪g',
  66. '碳水化合物g', '淀粉g', '总膳食纤维g', '胆固醇mg']
  67. KNOWN_CATS = ['主食', '乳制品', '干果', '坚果', '快餐', '水产品',
  68. '水果', '汤', '肉类', '蔬菜', '豆类及豆制品', '蛋类', '饮料']
  69. FOOD_SKIP_TEXTS = [
  70. '根据您的肠道菌群', '分值从-100', '食物推荐考虑', '食物推荐是综合',
  71. '需要注意的是', '本饮食推荐', '该饮食推荐根据', '后续表格中的营养',
  72. '16S 高通量测序', '基于机器学习和', '肠道菌群健康检测报告说明',
  73. '检测方法及局限性', '数据分析及模型', '结果解读及使用',
  74. '影响因素说明', '建议将检测结果', '营养建议说明',
  75. '重要提示', '推荐食物清单', '实际食用时需结合', '如有特殊疾病',
  76. '免责声明', '本检测报告仅供', '以上模型预测', '正常范围的定义',
  77. '募极生物',
  78. ]
  79. def norm(s):
  80. return ''.join(RADICAL_MAP.get(c, c) for c in s)
  81. def detect_format(lines):
  82. """检测 triplet / inline 格式"""
  83. text = '\n'.join(lines)
  84. has_triplet = '指标范围' in text and '疾病风险评估' in text
  85. for line in lines:
  86. if len(line) > 15 and re.search(r'[\u4e00-\u9fff]+[\d.]+[\u4e00-\u9fff/]+', line):
  87. for known in KNOWN_DISEASE_RISKS:
  88. if known in line:
  89. return 'inline'
  90. return 'triplet' if has_triplet else 'inline'
  91. def extract_text(file_path):
  92. """读取 PDF 并提取文本"""
  93. reader = PdfReader(file_path)
  94. lines = []
  95. for page in reader.pages:
  96. text = norm(page.extract_text() or '')
  97. for line in text.split('\n'):
  98. ls = line.strip()
  99. if ls:
  100. lines.append(ls)
  101. return lines
  102. def parse_overview(lines):
  103. """提取报告概述"""
  104. text = '\n'.join(lines)
  105. r = {}
  106. m = re.search(r'编号[::\s]*(\d+)', text)
  107. if m: r['report_number'] = m.group(1)
  108. m = re.search(r'姓名[::\s]*([\u4e00-\u9fff]{2,10})', text)
  109. if m: r['person_name'] = re.sub(r'(编号|年龄|性别|备注|肠道).*', '', m.group(1))[:4]
  110. m = re.search(r'年龄[::\s]*(\d+)', text)
  111. if m: r['age'] = int(m.group(1))
  112. m = re.search(r'性别[::\s]*([\u4e00-\u9fff])', text)
  113. if m: r['gender'] = 'male' if m.group(1) == '男' else 'female'
  114. for kw in ['健康总分', '菌群健康', '慢病控制', '营养均衡', '肠道菌群平衡',
  115. '菌群多样性', '有益菌', '有害菌', '核心菌属']:
  116. m = re.search(rf'{kw}\s*(\d+)', text)
  117. if m: r[kw] = int(m.group(1))
  118. m = re.search(r'肠道预测年龄[::\s]*([\d.]+)', text)
  119. if m: r['gut_age'] = m.group(1)
  120. m = re.search(r'肠型[::\s]*(\S+)', text)
  121. if m: r['gut_type'] = m.group(1)
  122. return r
  123. def parse_triplet_until(lines, stop_markers):
  124. """三元组解析:3行一组 名称/数值/状态"""
  125. results = []
  126. i = 0
  127. while i < len(lines):
  128. if any(lines[i] == sm or lines[i].startswith(sm) for sm in stop_markers):
  129. break
  130. if lines[i] in ('指标范围', '名称', '丰度', '评估'):
  131. i += 1
  132. continue
  133. name = lines[i]
  134. if i + 2 >= len(lines): break
  135. val = lines[i + 1]
  136. status = lines[i + 2]
  137. if re.match(r'^-?\d+\.?\d*$', val):
  138. results.append({'name': name, 'value': val, 'status': status})
  139. i += 3
  140. else:
  141. i += 1
  142. return results
  143. def parse_report_pdf(file_path: str) -> dict:
  144. """主函数:解析 PDF 返回结构化数据"""
  145. lines = extract_text(file_path)
  146. fmt = detect_format(lines)
  147. result = {'format': fmt, 'overview': parse_overview(lines)}
  148. # 疾病风险评估
  149. for i, line in enumerate(lines):
  150. if '疾病风险评估' in line and '注' not in line:
  151. risks = parse_triplet_until(lines[i+1:],
  152. ['主要营养评估', '氨基酸评估', '维生素评估', '微量元素评估', '抗生素风险评估'])
  153. result['disease_risks'] = [r for r in risks if '注' not in r['name']]
  154. break
  155. # 主要营养评估
  156. for i, line in enumerate(lines):
  157. if '主要营养评估' in line:
  158. nutrients = parse_triplet_until(lines[i+1:], ['氨基酸评估'])
  159. result['nutrition'] = nutrients[:5]
  160. break
  161. # 氨基酸评估
  162. for i, line in enumerate(lines):
  163. if '氨基酸评估' in line:
  164. aminos = parse_triplet_until(lines[i+1:], ['维生素评估', '微量元素评估'])
  165. result['amino_acids'] = aminos
  166. break
  167. # 维生素评估
  168. for i, line in enumerate(lines):
  169. if '维生素评估' in line:
  170. vits = parse_triplet_until(lines[i+1:], ['微量元素评估', '抗生素风险评估'])
  171. result['vitamins'] = [r for r in vits if '维生素' in r['name']]
  172. result['trace_elements'] = [r for r in vits if '维生素' not in r['name']]
  173. break
  174. # 菌群检出详细列表(核心/益生菌/有害菌/病原菌 + 门纲目科属种)
  175. bacteria_tables = _extract_bacteria_tables(file_path)
  176. result['菌群检出详细列表'] = bacteria_tables
  177. # 个体化食物推荐表(保留原始键名与结构)
  178. food_rows, food_fmt = _extract_food_rows(file_path)
  179. result['个体化食物推荐表'] = {
  180. '格式': food_fmt,
  181. '条目数': len(food_rows),
  182. '数据': food_rows
  183. }
  184. return result
  185. def parse_report_pdf_with_fallback(file_path: str) -> dict:
  186. """算法解析 + 简单校验,返回结构化数据"""
  187. result = parse_report_pdf(file_path)
  188. # 简单校验:如果关键字段缺失,标记为解析不完整
  189. if not result.get('overview', {}).get('overallScore') and \
  190. not result.get('overview', {}).get('健康总分'):
  191. result['_parse_incomplete'] = True
  192. return result
  193. def _parse_bacteria_table(reader, pages_text, full_text, title, fmt, skip_header=False):
  194. """从PDF中解析一个菌群表格(移植自 extract_full_report_v5.py)"""
  195. results = []
  196. sidx = full_text.find(title)
  197. if sidx == -1:
  198. return results
  199. # 病原菌检出表特殊处理:找"仅列出检出的病原菌"(跳过前面的说明文字中的"病原菌")
  200. if title == '病原菌':
  201. better_sidx = full_text.find('仅列出检出的病原菌')
  202. if better_sidx != -1:
  203. sidx = better_sidx
  204. # 找表格结束位置(下一个标题或页尾)
  205. end_pos = len(full_text)
  206. for t in (BACTERIA_TABLE_TITLES + PHYLUM_TABLE_TITLES + CLASS_TABLE_TITLES
  207. + ORDER_TABLE_TITLES + FAMILY_TABLE_TITLES + GENUS_TABLE_TITLES
  208. + SPECIES_TABLE_TITLES + DISEASE_BACTERIA_TITLES
  209. + ['指标范围', '个体化食物推荐表', '报告总结', '健康总分']):
  210. if t == title:
  211. continue
  212. ei = full_text.find(t, sidx + len(title))
  213. if ei != -1 and ei < end_pos:
  214. end_pos = ei
  215. region = full_text[sidx:end_pos]
  216. # 检测区域实际格式:如果换行数很少(<3)则是inline格式,即使全局fmt=triplet
  217. lines_from_region = [l.strip() for l in region.split('\n') if l.strip()]
  218. actual_fmt = fmt
  219. if fmt == 'triplet' and len(lines_from_region) <= 5:
  220. actual_fmt = 'inline'
  221. if actual_fmt == 'triplet':
  222. # 三元组格式:每个字段单独一行
  223. lines = [l.strip() for l in region.split('\n') if l.strip()]
  224. start = 0
  225. for i, line in enumerate(lines):
  226. if line == '名称':
  227. start = i + 1
  228. break
  229. if line.startswith('名称'):
  230. if skip_header:
  231. start = i + 1
  232. break
  233. i = start
  234. while i < len(lines):
  235. name = lines[i]
  236. if not name or len(name) <= 1 or name in ['说明', '检测结果', '结果解释', '建议']:
  237. i += 1
  238. continue
  239. if name.startswith('说明:') or name.startswith('改善方式'):
  240. i += 1
  241. continue
  242. if len(name) > 80:
  243. i += 1
  244. continue
  245. # 找丰度%
  246. if i + 1 < len(lines) and re.match(r'^[\d]+\.?[\d]*%?$|^ND$', lines[i + 1]):
  247. pct = lines[i + 1]
  248. normal_range = ''
  249. pop_level = ''
  250. detection_rate = ''
  251. desc = ''
  252. j = i + 2
  253. # 正常范围 (允许小数,如 0.06-6.96, 0.03-3.07)
  254. if j < len(lines) and re.match(r'^[\d]+\.?[\d]*-[\d]+\.?[\d]*$', lines[j]):
  255. normal_range = lines[j]
  256. j += 1
  257. # 人群水平% (允许小数,如 15.66%)
  258. if j < len(lines) and re.match(r'^\d+\.?\d*%$', lines[j]):
  259. pop_level = lines[j]
  260. j += 1
  261. # 检出率%
  262. if j < len(lines) and re.match(r'^\d+\.?\d*%$', lines[j]):
  263. detection_rate = lines[j]
  264. j += 1
  265. # 说明
  266. if j < len(lines) and lines[j].startswith('说明'):
  267. desc = lines[j]
  268. j += 1
  269. # 改善方式
  270. if j < len(lines) and lines[j].startswith('改善方式'):
  271. if desc:
  272. desc += ' | ' + lines[j]
  273. else:
  274. desc = lines[j]
  275. j += 1
  276. entry = {'名称': name, '丰度%': pct}
  277. if normal_range:
  278. entry['正常范围%'] = normal_range
  279. if pop_level:
  280. entry['人群水平%'] = pop_level
  281. if detection_rate:
  282. entry['检出率%'] = detection_rate
  283. if desc:
  284. entry['说明'] = desc
  285. results.append(entry)
  286. i = j
  287. else:
  288. i += 1
  289. else:
  290. # inline 格式
  291. sample = region[:500]
  292. truly_compressed = bool(re.search(r'[a-z]\d', sample, re.IGNORECASE))
  293. if not truly_compressed:
  294. # 有空格分隔的inline格式,使用re.finditer
  295. region_clean = region
  296. for hdr in ['名称', '丰度%', '正常范围%', '处于人群%水平', '%正常人有检出',
  297. '人群水平%', '%人检出']:
  298. region_clean = region_clean.replace(hdr, '')
  299. region_clean = re.sub(
  300. r'说明:[\u4e00-\u9fff\s,。、;:,.;:()()、/a-zA-Z0-9\-]{10,}?(?=[\u4e00-\u9fff]|$)',
  301. '', region_clean)
  302. # 扫描所有匹配的数据行
  303. for m in re.finditer(
  304. r'([\u4e00-\u9fff]{2,12}(?:[((][\u4e00-\u9fff\w]+[))])?)\s+' # 中文名
  305. r'(?:[A-Z][a-z]+(?:\s[A-Z][a-z]+)*\s+)?' # 可选英文名
  306. r'(ND|[\d]+\.?[\d]*%?)\s+' # 丰度
  307. r'([\d]+\.?[\d]*-[\d]+\.?[\d]*)?\s*' # 可选正常范围
  308. r'(\d+\.?\d*%?)\s+' # 人群水平%
  309. r'(\d+\.?\d*%)', # 检出率%
  310. region_clean):
  311. name = m.group(1).strip()
  312. pct = m.group(2)
  313. if '病原菌' in name or '构成表' in name or '说明' in name or len(name) <= 1:
  314. continue
  315. normal_range = m.group(3) or ''
  316. pop_level = m.group(4)
  317. if not pop_level.endswith('%'):
  318. pop_level += '%'
  319. detection_rate = m.group(5)
  320. entry = {'名称': name, '丰度%': pct}
  321. if normal_range:
  322. entry['正常范围%'] = normal_range
  323. entry['人群水平%'] = pop_level
  324. entry['检出率%'] = detection_rate
  325. results.append(entry)
  326. # 模式1没有匹配时:仅中文名+丰度+人群水平(+检出率)
  327. if not results:
  328. for m in re.finditer(
  329. r'([\u4e00-\u9fff]{2,10}[\u4e00-\u9fff]?)\s+'
  330. r'(ND|[\d]+\.?[\d]*%?)\s+'
  331. r'(\d+\.?\d*%)\s+'
  332. r'(\d+\.?\d*%)?',
  333. region_clean):
  334. name = m.group(1).strip()
  335. pct = m.group(2)
  336. if '病原菌' in name or '构成表' in name or '说明' in name or len(name) <= 1:
  337. continue
  338. pop_level = m.group(3)
  339. detection_rate = m.group(4) or ''
  340. entry = {'名称': name, '丰度%': pct, '人群水平%': pop_level}
  341. if detection_rate:
  342. entry['检出率%'] = detection_rate
  343. results.append(entry)
  344. return results
  345. def _parse_phylum_tables(reader, full_text, fmt, title_list=None):
  346. """提取菌群层级构成表(门/纲/目/科/属/种 level)(移植自 extract_full_report_v5.py)"""
  347. if title_list is None:
  348. title_list = PHYLUM_TABLE_TITLES
  349. results = []
  350. for phylum_title in title_list:
  351. rows = _parse_bacteria_table(reader, None, full_text, phylum_title, fmt)
  352. results.extend(rows)
  353. return results
  354. def _parse_taxonomy_levels(reader, full_text, fmt):
  355. """从"菌群检出详细列表"中提取纲目科属种各级数据(移植自 extract_full_report_v5.py)"""
  356. results = {}
  357. sidx = full_text.find('菌群检出详细列表')
  358. if sidx == -1:
  359. return results
  360. end_pos = len(full_text)
  361. for t in ['个体化食物推荐表', '报告总结', '健康总分']:
  362. ei = full_text.find(t, sidx)
  363. if ei != -1 and ei < end_pos:
  364. end_pos = ei
  365. section = full_text[sidx:end_pos]
  366. for level, level_name in [('纲', '菌纲构成'), ('目', '菌目构成'),
  367. ('科', '菌科构成'), ('属', '菌属构成'),
  368. ('种', '菌种构成')]:
  369. marker = f'\n{level}\n名称\n丰度%'
  370. marker2 = f'{level} 名称 丰度%'
  371. li = section.find(marker)
  372. level_start = None
  373. if li == -1:
  374. li2 = section.find(marker2)
  375. if li2 != -1:
  376. li = li2
  377. level_start = li2 + len(marker2)
  378. else:
  379. compressed_marker = f'{level}名称丰度%人群水平%%人检出'
  380. cli = section.find(compressed_marker)
  381. if cli == -1:
  382. continue
  383. # 压缩格式解析:用正则提取数据
  384. level_start = cli + len(compressed_marker)
  385. level_end = len(section)
  386. for next_level in ['目', '科', '属', '种']:
  387. if next_level == level:
  388. continue
  389. ni = section.find(f'{next_level}名称丰度%人群水平%%人检出', level_start)
  390. if ni != -1 and ni < level_end:
  391. level_end = ni
  392. break
  393. level_region = section[level_start:level_end]
  394. compressed_pattern = re.compile(
  395. r'([\u4e00-\u9fff·]+(?:\s[\u4e00-\u9fff·]+)?\s+)?' # 可选中文名
  396. r'([A-Za-z][A-Za-z\s.\-]*?)' # 拉丁名(可能含空格)
  397. r'(\d+\.?\d*%)(\d+\.?\d*%)(\d+\.?\d*%)' # 三连百分比
  398. )
  399. rows = []
  400. for m in compressed_pattern.finditer(level_region):
  401. cn_name = (m.group(1) or '').strip()
  402. latin_name = m.group(2).strip()
  403. pct = m.group(3)
  404. pop_level = m.group(4)
  405. detection = m.group(5)
  406. name = cn_name if cn_name else latin_name
  407. entry = {'名称': name, '丰度%': pct, '人群水平%': pop_level, '检出率%': detection}
  408. rows.append(entry)
  409. if rows:
  410. results[level_name] = rows
  411. continue
  412. else:
  413. level_start = li + len(marker)
  414. level_end = len(section)
  415. for next_level in ['纲', '目', '科', '属', '种']:
  416. if next_level == level:
  417. continue
  418. ni = section.find(f'\n{next_level}\n名称', level_start)
  419. if ni != -1 and ni < level_end:
  420. level_end = ni
  421. break
  422. level_region = section[level_start:level_end]
  423. lines = [l.strip() for l in level_region.split('\n') if l.strip()]
  424. rows = []
  425. i = 0
  426. while i < len(lines):
  427. if lines[i] in ['名称', '丰度%', '人群水平%', '%人检出']:
  428. i += 1
  429. continue
  430. name = lines[i]
  431. if i + 2 < len(lines) and re.match(r'^[\d]+\.?[\d]*%?$', lines[i + 1]):
  432. pct = lines[i + 1]
  433. pop_level = lines[i + 2] if i + 2 < len(lines) else ''
  434. detection = lines[i + 3] if i + 3 < len(lines) and re.match(r'^[\d.]+%$', lines[i + 3]) else ''
  435. entry = {'名称': name, '丰度%': pct, '人群水平%': pop_level}
  436. if detection:
  437. entry['检出率%'] = detection
  438. rows.append(entry)
  439. i += 4 if detection else 3
  440. else:
  441. i += 1
  442. if rows:
  443. results[level_name] = rows
  444. return results
  445. def _extract_bacteria_tables(pdf_path):
  446. """提取菌群检出详细列表,返回 {中文分组名: [行]}(移植自 extract_full_report_v5.py)
  447. 与原始脚本保持一致的文本构造与格式检测:full_text 使用
  448. '\\n'.join(norm(p.extract_text()) for p in reader.pages)(不 strip、不过滤空行),
  449. fmt 使用原脚本 L672-683 的独立检测逻辑。
  450. """
  451. reader = PdfReader(pdf_path)
  452. full_text = '\n'.join(norm(p.extract_text()) for p in reader.pages)
  453. # 原脚本 extract_bacteria_tables 的格式检测(L674-683)
  454. fmt = 'triplet' if '指标范围' in full_text and '疾病风险评估' in full_text else 'inline'
  455. for pt in [p.extract_text() for p in reader.pages]:
  456. t = norm(pt)
  457. if '疾病风险评估' in t and '指标范围' in t:
  458. for line in t.split('\n'):
  459. if re.search(r'[\u4e00-\u9fff]+\d+\.?\d*[\u4e00-\u9fff]+', line.strip()):
  460. fmt = 'inline'
  461. break
  462. break
  463. all_tables = {}
  464. # 核心菌属构成表1-3
  465. core_genus = []
  466. for i in range(1, 4):
  467. title = f'核心菌属构成表{i}'
  468. rows = _parse_bacteria_table(reader, None, full_text, title, fmt)
  469. core_genus.extend(rows)
  470. all_tables['核心菌属'] = core_genus
  471. # 益生菌(使用更精确的表头定位,跳过前面的说明文字)
  472. prob_marker = '仅列出丰度前22的益生菌种'
  473. prob_sidx = full_text.find(prob_marker)
  474. if prob_sidx != -1:
  475. prob_rows = _parse_bacteria_table(reader, None, full_text, prob_marker, fmt, skip_header=True)
  476. else:
  477. prob_rows = _parse_bacteria_table(reader, None, full_text, '益生菌', fmt, skip_header=True)
  478. all_tables['益生菌'] = [r for r in prob_rows if r.get('名称') and r['名称'] not in
  479. ['我的益生菌都为ND', '仅列出丰度前22的益生菌种。']]
  480. # 有害菌属构成表1-2
  481. harmful = []
  482. for i in range(1, 3):
  483. title = f'有害菌属构成表{i}'
  484. rows = _parse_bacteria_table(reader, None, full_text, title, fmt)
  485. harmful.extend(rows)
  486. all_tables['有害菌属'] = harmful
  487. # 其它重要菌属
  488. other_rows = _parse_bacteria_table(reader, None, full_text, '其它重要菌属构成表', fmt)
  489. all_tables['其它重要菌属'] = other_rows
  490. # 病原菌属构成表
  491. patho_genus = _parse_bacteria_table(reader, None, full_text, '病原菌属构成表', fmt)
  492. all_tables['病原菌属'] = patho_genus
  493. # 病原菌(检出列表)
  494. patho_rows = _parse_bacteria_table(reader, None, full_text, '病原菌', fmt, skip_header=True)
  495. all_tables['病原菌检出'] = [r for r in patho_rows if r.get('名称') and len(r['名称']) >= 2
  496. and '仅列出' not in r['名称'] and '说明' not in r['名称']]
  497. # 菌门构成表(phylum level)
  498. phylum_rows = _parse_phylum_tables(reader, full_text, fmt)
  499. if phylum_rows:
  500. all_tables['菌门构成'] = phylum_rows
  501. # 菌纲构成表
  502. class_rows = _parse_phylum_tables(reader, full_text, fmt, CLASS_TABLE_TITLES)
  503. if class_rows:
  504. all_tables['菌纲构成'] = class_rows
  505. # 菌目构成表
  506. order_rows = _parse_phylum_tables(reader, full_text, fmt, ORDER_TABLE_TITLES)
  507. if order_rows:
  508. all_tables['菌目构成'] = order_rows
  509. # 菌科构成表
  510. family_rows = _parse_phylum_tables(reader, full_text, fmt, FAMILY_TABLE_TITLES)
  511. if family_rows:
  512. all_tables['菌科构成'] = family_rows
  513. # 菌属构成表
  514. genus_rows = _parse_phylum_tables(reader, full_text, fmt, GENUS_TABLE_TITLES)
  515. if genus_rows:
  516. all_tables['菌属构成'] = genus_rows
  517. # 菌种构成表
  518. species_rows = _parse_phylum_tables(reader, full_text, fmt, SPECIES_TABLE_TITLES)
  519. if species_rows:
  520. all_tables['菌种构成'] = species_rows
  521. # 菌群层级(纲目科属种)- 从"菌群检出详细列表"统一入口提取
  522. taxonomy_rows = _parse_taxonomy_levels(reader, full_text, fmt)
  523. for key, rows in taxonomy_rows.items():
  524. if rows:
  525. all_tables[key] = rows
  526. return all_tables
  527. def split_7_fields(s):
  528. """将压缩数字串切分为 7 个字段(移植自 extract_full_report_v5.py)"""
  529. results = []
  530. ranges = [(2, 4), (1, 2), (1, 2), (1, 2), (1, 2), (1, 2), (1, 4)]
  531. def backtrack(pos, idx, nums):
  532. if idx == 7:
  533. if pos == len(s):
  534. results.append(list(nums))
  535. return
  536. if pos >= len(s):
  537. return
  538. lo, hi = ranges[idx]
  539. for w in range(lo, min(hi + 1, len(s) - pos + 1)):
  540. chunk = s[pos:pos + w]
  541. if chunk.isdigit():
  542. backtrack(pos + w, idx + 1, nums + [int(chunk)])
  543. backtrack(0, 0, [])
  544. return results
  545. def decode_compressed(name, num_str, ref_vals=None):
  546. """解码压缩食物数字串(移植自 extract_full_report_v5.py)"""
  547. raw = num_str.lstrip('-')
  548. has_neg = num_str.startswith('-')
  549. candidates = []
  550. for rec_len in range(1, 3):
  551. if rec_len > len(raw):
  552. continue
  553. rec = ('-' if has_neg else '') + raw[:rec_len]
  554. try:
  555. rec_val = int(rec)
  556. if not (-100 <= rec_val <= 100):
  557. continue
  558. except Exception:
  559. continue
  560. remain = raw[rec_len:]
  561. for nums in split_7_fields(remain):
  562. if ref_vals:
  563. matches = sum(1 for i in range(7) if ref_vals[i] == nums[i])
  564. if matches >= 6:
  565. candidates.append([rec_val] + nums)
  566. else:
  567. candidates.append([rec_val] + nums)
  568. if not candidates:
  569. return None
  570. if ref_vals:
  571. candidates.sort(key=lambda r: (sum(1 for i in range(7) if ref_vals[i] == r[1:][i]),
  572. -len(str(abs(r[0])))), reverse=True)
  573. if sum(1 for i in range(7) if ref_vals[i] == candidates[0][1:][i]) < 6:
  574. return None
  575. else:
  576. candidates.sort(key=lambda r: (len(str(abs(r[0]))),
  577. -sum(1 for i in range(7) if r[1:][i] == 0)))
  578. return candidates[0]
  579. def extract_food_table(pdf_path, ref_lookup=None):
  580. """提取个体化食物推荐表(移植自 extract_full_report_v5.py)"""
  581. reader = PdfReader(pdf_path)
  582. food_start = None
  583. for i, page in enumerate(reader.pages):
  584. if '个体化食物推荐表' in page.extract_text():
  585. food_start = i
  586. break
  587. if food_start is None:
  588. return [], 'not_found'
  589. first_text = norm(reader.pages[food_start + 1].extract_text())
  590. lines = [l.strip() for l in first_text.split('\n')
  591. if l.strip() and not re.match(r'\d+/\d+', l)]
  592. # 判断压缩格式:任何一行超过100字符(单行密集格式),或前10行中超过3行长行
  593. is_compressed = (len(lines) >= 1 and any(len(l) > 100 for l in lines[:10])) or \
  594. sum(1 for l in lines[:10] if len(l) > 100) >= 2
  595. rows = []
  596. if is_compressed:
  597. fmt = 'compressed'
  598. for i in range(food_start + 1, len(reader.pages)):
  599. text = norm(reader.pages[i].extract_text())
  600. text = re.sub(r'\d+/\d+', '', text)
  601. header = '名称分类推荐指数能量KJ蛋白g脂肪g碳水化合物g淀粉g总膳食纤维g胆固醇mg'
  602. text = text.replace(header, '')
  603. for kw in FOOD_SKIP_TEXTS:
  604. text = text.replace(kw, '')
  605. while text:
  606. best_cat, best_idx = None, len(text)
  607. for cat in KNOWN_CATS:
  608. idx = text.find(cat)
  609. if idx != -1 and idx < best_idx:
  610. best_idx, best_cat = idx, cat
  611. if best_cat is None:
  612. break
  613. name = text[:best_idx]
  614. text = text[best_idx + len(best_cat):]
  615. num_str = ''
  616. while text and (text[0].isdigit() or text[0] in '-\u2212\u2014'):
  617. c = '-' if text[0] in '\u2212\u2014' else text[0]
  618. num_str += c
  619. text = text[1:]
  620. if not name or not num_str:
  621. continue
  622. ref_vals = ref_lookup.get(name) if ref_lookup else None
  623. decoded = decode_compressed(name, num_str, ref_vals)
  624. if decoded:
  625. rows.append(dict(zip(COLUMNS_FOOD, [name, best_cat] + [str(v) for v in decoded])))
  626. else:
  627. fmt = 'vertical'
  628. all_lines = []
  629. for i in range(food_start + 1, len(reader.pages)):
  630. for line in norm(reader.pages[i].extract_text()).split('\n'):
  631. lt = line.strip()
  632. if not lt or re.match(r'\d+/\d+', lt) or lt in COLUMNS_FOOD:
  633. continue
  634. if len(lt) > 60 and any(k in lt for k in FOOD_SKIP_TEXTS):
  635. continue
  636. all_lines.append(lt)
  637. i = 0
  638. while i + 9 < len(all_lines):
  639. name = all_lines[i].strip()
  640. cat = all_lines[i + 1].strip()
  641. if cat not in KNOWN_CATS:
  642. i += 1
  643. continue
  644. nums = []
  645. ok = True
  646. for j in range(2, 10):
  647. v = all_lines[i + j].replace('\u2212', '-').replace('\u2014', '-').strip()
  648. try:
  649. int(v)
  650. nums.append(v)
  651. except Exception:
  652. ok = False
  653. break
  654. if ok and len(nums) == 8:
  655. rows.append(dict(zip(COLUMNS_FOOD, [name, cat] + nums)))
  656. i += 1
  657. return rows, fmt
  658. def _extract_food_rows(pdf_path):
  659. """食物推荐表解析:优先复用同目录其他报告的参考营养表做压缩格式解码"""
  660. ref_nutrition = {}
  661. base_dir = os.path.dirname(pdf_path) or '.'
  662. for fname in sorted(os.listdir(base_dir)):
  663. if fname.lower().endswith('.pdf') and fname != os.path.basename(pdf_path):
  664. try:
  665. tr, _ = extract_food_table(os.path.join(base_dir, fname))
  666. if len(tr) > 100:
  667. for r in tr:
  668. ref_nutrition[r['名称']] = [int(r[k]) for k in
  669. ['能量KJ', '蛋白g', '脂肪g', '碳水化合物g', '淀粉g',
  670. '总膳食纤维g', '胆固醇mg']]
  671. break
  672. except Exception:
  673. continue
  674. food_rows, food_fmt = extract_food_table(pdf_path, ref_nutrition or None)
  675. return food_rows, food_fmt