| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899 |
- # -*- coding: utf-8 -*-
- """
- Process remaining PDFs - analyze content to identify type and add proper prefix
- """
- import os
- import re
- import fitz
- from type_detector import detect_type
- # Known patterns that already have prefix
- ALREADY_PREFIXED = re.compile(r'^(A1_|A2_|B[0-9]_|[A-Z][0-9]\s)')
- def identify_report_type(pdf_path):
- """Identify report type by reading PDF content (fingerprint-based)"""
- try:
- doc = fitz.open(pdf_path)
- text = ""
- for page in doc[:5]:
- text += page.get_text()
- doc.close()
-
- report_type, scores = detect_type(text, pdf_path)
- return report_type
- except Exception as e:
- print(f"Error reading {pdf_path}: {e}")
- return 'Error'
- def process_files():
- # 自动计算项目根目录
- script_dir = os.path.dirname(os.path.abspath(__file__))
- project_dir = os.path.dirname(os.path.dirname(script_dir))
- reports_dir = os.path.join(project_dir, "报告")
- output_dir = os.path.join(project_dir, "报告")
-
- # Get all PDF files
- all_files = [f for f in os.listdir(reports_dir) if f.endswith('.pdf')]
-
- # Filter to unprocessed files (no proper prefix)
- unprocessed = []
- for f in all_files:
- if not ALREADY_PREFIXED.match(f):
- unprocessed.append(f)
-
- print(f"Total files: {len(all_files)}")
- print(f"Already processed (has prefix): {len(all_files) - len(unprocessed)}")
- print(f"Need to process: {len(unprocessed)}")
-
- # Process in batches
- results = {'A1': [], 'A2': [], 'B2': [], 'B3': [], 'B4': [], 'B5': [], 'B6': [], 'C1': [], 'Unknown': [], 'Error': []}
-
- for i, filename in enumerate(unprocessed):
- if i % 50 == 0:
- print(f"Processing {i}/{len(unprocessed)}...")
-
- pdf_path = os.path.join(reports_dir, filename)
- report_type = identify_report_type(pdf_path)
- results[report_type].append(filename)
-
- # Extract name from filename for renaming
- name_match = re.search(r'[_-](\w+)[_-]\d{13,}', filename)
- if name_match:
- name = name_match.group(1)
- else:
- # Try alternative pattern
- parts = re.split(r'[_\-]', filename)
- name = parts[-2] if len(parts) >= 2 else 'unknown'
-
- # Create new filename with proper prefix
- ext = '.pdf'
- # Get name part (before timestamp)
- name_match = re.search(r'(.+?)_\d{13,}', filename)
- if name_match:
- base_name = name_match.group(1).strip()
- new_filename = f"{report_type}_{base_name}{ext}"
- else:
- new_filename = f"{report_type}_{filename}"
-
- # Rename if not already correct
- if not filename.startswith(report_type + '_'):
- try:
- new_path = os.path.join(reports_dir, new_filename)
- # Check if target exists
- if not os.path.exists(new_path):
- os.rename(pdf_path, new_path)
- print(f"Renamed: {filename} -> {new_filename}")
- else:
- print(f"Skipped (exists): {filename}")
- except Exception as e:
- print(f"Error renaming {filename}: {e}")
-
- # Print summary
- print("\n=== Summary ===")
- for t, files in results.items():
- print(f"{t}: {len(files)}")
-
- return results
- if __name__ == "__main__":
- process_files()
|