Duplicate File Finder
This utility scans a directory tree to identify duplicate files based on their content (hash), not just their names. It uses hashlib to generate a unique fingerprint for each file.
Modules Used:
- hashlib: To calculate secure hashes (SHA-256) of file contents.
- pathlib: To recursively traverse directories.
- argparse: To handle command-line arguments.
The Code
Save this as dedupe.py.
import hashlib
import argparse
from pathlib import Path
from collections import defaultdict
def get_file_hash(file_path, block_size=65536):
"""Calculates the SHA-256 hash of a file."""
sha256 = hashlib.sha256()
try:
with open(file_path, 'rb') as f:
while True:
data = f.read(block_size)
if not data:
break
sha256.update(data)
return sha256.hexdigest()
except OSError as e:
print(f"Error reading {file_path}: {e}")
return None
def find_duplicates(directory):
hashes = defaultdict(list)
path = Path(directory)
if not path.exists():
print(f"Error: Directory '{directory}' not found.")
return
print(f"Scanning '{directory}' for duplicates...")
# Find all files recursively
files = [f for f in path.rglob("*") if f.is_file()]
for i, file_path in enumerate(files):
file_hash = get_file_hash(file_path)
if file_hash:
hashes[file_hash].append(file_path)
return hashes
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Duplicate File Finder")
parser.add_argument("directory", help="Directory to scan")
parser.add_argument("--delete", action="store_true", help="Delete duplicates (keep one)")
args = parser.parse_args()
results = find_duplicates(args.directory)
# Filter for hashes with more than one file
duplicates = {k: v for k, v in results.items() if len(v) > 1}
if not duplicates:
print("No duplicates found.")
else:
print(f"Found {len(duplicates)} sets of duplicates:")
for file_hash, paths in duplicates.items():
print(f"\nHash: {file_hash[:8]}...")
# The first file is the "original" (or just the first one found)
original = paths[0]
print(f" Keep: {original}")
for duplicate in paths[1:]:
if args.delete:
try:
# duplicate.unlink() # Uncomment to actually delete
print(f" Deleted: {duplicate} (Simulated)")
except OSError as e:
print(f" Error deleting {duplicate}: {e}")
else:
print(f" Duplicate: {duplicate}")
Usage
# Scan for duplicates
python dedupe.py ./photos
# Scan and simulate deletion of duplicates
python dedupe.py ./downloads --delete