| 12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364 |
- #!/usr/bin/env python
- # -*- coding: utf-8 -*-
- """
- 检查报告中所有A1_开头的文件,如果同目录中有内容完全相同的文件,则删掉这个A1_开头的文件
- """
- import os
- import hashlib
- from pathlib import Path
- def file_hash(filepath):
- """计算文件 SHA256 哈希"""
- h = hashlib.sha256()
- with open(filepath, 'rb') as f:
- for chunk in iter(lambda: f.read(65536), b''):
- h.update(chunk)
- return h.hexdigest()
- def main():
- # 自动计算项目根目录 (scripts/reports/ → 项目根)
- report_dir = Path(__file__).parent.parent.parent / "报告"
-
- # 找出所有 A1_ 开头的文件
- a1_files = list(report_dir.glob("A1_*.pdf"))
- print(f"找到 {len(a1_files)} 个 A1_ 开头的文件")
-
- # 计算所有文件的哈希
- all_files = list(report_dir.glob("*.pdf"))
- hash_map = {}
-
- for f in all_files:
- h = file_hash(str(f))
- if h not in hash_map:
- hash_map[h] = []
- hash_map[h].append(f)
-
- # 找出重复的文件
- deleted_count = 0
- for h, files in hash_map.items():
- if len(files) > 1:
- a1_dups = [f for f in files if f.name.startswith("A1_")]
- non_a1 = [f for f in files if not f.name.startswith("A1_")]
-
- if a1_dups and non_a1:
- for f in a1_dups:
- try:
- f.unlink()
- print(f"[DELETED] {f.name} (与 {non_a1[0].name} 内容相同)")
- deleted_count += 1
- except Exception as e:
- print(f"[FAIL] {f.name}: {e}")
- elif len(a1_dups) > 1:
- keep = a1_dups[0]
- for f in a1_dups[1:]:
- try:
- f.unlink()
- print(f"[DELETED] {f.name} (与 {keep.name} 内容相同)")
- deleted_count += 1
- except Exception as e:
- print(f"[FAIL] {f.name}: {e}")
- print(f"\n总共删除了 {deleted_count} 个重复的 A1_ 文件")
- if __name__ == "__main__":
- main()
|