process_all_pdfs.py 3.4 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899
  1. # -*- coding: utf-8 -*-
  2. """
  3. Process remaining PDFs - analyze content to identify type and add proper prefix
  4. """
  5. import os
  6. import re
  7. import fitz
  8. from type_detector import detect_type
  9. # Known patterns that already have prefix
  10. ALREADY_PREFIXED = re.compile(r'^(A1_|A2_|B[0-9]_|[A-Z][0-9]\s)')
  11. def identify_report_type(pdf_path):
  12. """Identify report type by reading PDF content (fingerprint-based)"""
  13. try:
  14. doc = fitz.open(pdf_path)
  15. text = ""
  16. for page in doc[:5]:
  17. text += page.get_text()
  18. doc.close()
  19. report_type, scores = detect_type(text, pdf_path)
  20. return report_type
  21. except Exception as e:
  22. print(f"Error reading {pdf_path}: {e}")
  23. return 'Error'
  24. def process_files():
  25. # 自动计算项目根目录
  26. script_dir = os.path.dirname(os.path.abspath(__file__))
  27. project_dir = os.path.dirname(os.path.dirname(script_dir))
  28. reports_dir = os.path.join(project_dir, "报告")
  29. output_dir = os.path.join(project_dir, "报告")
  30. # Get all PDF files
  31. all_files = [f for f in os.listdir(reports_dir) if f.endswith('.pdf')]
  32. # Filter to unprocessed files (no proper prefix)
  33. unprocessed = []
  34. for f in all_files:
  35. if not ALREADY_PREFIXED.match(f):
  36. unprocessed.append(f)
  37. print(f"Total files: {len(all_files)}")
  38. print(f"Already processed (has prefix): {len(all_files) - len(unprocessed)}")
  39. print(f"Need to process: {len(unprocessed)}")
  40. # Process in batches
  41. results = {'A1': [], 'A2': [], 'B2': [], 'B3': [], 'B4': [], 'B5': [], 'B6': [], 'C1': [], 'Unknown': [], 'Error': []}
  42. for i, filename in enumerate(unprocessed):
  43. if i % 50 == 0:
  44. print(f"Processing {i}/{len(unprocessed)}...")
  45. pdf_path = os.path.join(reports_dir, filename)
  46. report_type = identify_report_type(pdf_path)
  47. results[report_type].append(filename)
  48. # Extract name from filename for renaming
  49. name_match = re.search(r'[_-](\w+)[_-]\d{13,}', filename)
  50. if name_match:
  51. name = name_match.group(1)
  52. else:
  53. # Try alternative pattern
  54. parts = re.split(r'[_\-]', filename)
  55. name = parts[-2] if len(parts) >= 2 else 'unknown'
  56. # Create new filename with proper prefix
  57. ext = '.pdf'
  58. # Get name part (before timestamp)
  59. name_match = re.search(r'(.+?)_\d{13,}', filename)
  60. if name_match:
  61. base_name = name_match.group(1).strip()
  62. new_filename = f"{report_type}_{base_name}{ext}"
  63. else:
  64. new_filename = f"{report_type}_{filename}"
  65. # Rename if not already correct
  66. if not filename.startswith(report_type + '_'):
  67. try:
  68. new_path = os.path.join(reports_dir, new_filename)
  69. # Check if target exists
  70. if not os.path.exists(new_path):
  71. os.rename(pdf_path, new_path)
  72. print(f"Renamed: {filename} -> {new_filename}")
  73. else:
  74. print(f"Skipped (exists): {filename}")
  75. except Exception as e:
  76. print(f"Error renaming {filename}: {e}")
  77. # Print summary
  78. print("\n=== Summary ===")
  79. for t, files in results.items():
  80. print(f"{t}: {len(files)}")
  81. return results
  82. if __name__ == "__main__":
  83. process_files()