updates and backlog
This commit is contained in:
@@ -0,0 +1,140 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
import os
|
||||
import sys
|
||||
import subprocess
|
||||
import re
|
||||
import shutil
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Optional, Dict, Set
|
||||
|
||||
def check_dependencies() -> bool:
|
||||
"""Check if required commands are available."""
|
||||
for cmd in ['file', 'dctfilename']:
|
||||
if not shutil.which(cmd):
|
||||
print(f"Error: {cmd} command not found")
|
||||
return False
|
||||
return True
|
||||
|
||||
def is_processed_file(filename: str) -> bool:
|
||||
"""Check if filename matches our format (NNNN_HASH.ext)."""
|
||||
pattern = r'^[0-9]{4}_[0-9a-f]{16}\.(jpg|jpeg|png)$'
|
||||
return bool(re.match(pattern, filename.lower()))
|
||||
|
||||
def extract_hash(filename: str) -> Optional[str]:
|
||||
"""Extract hash from processed filename."""
|
||||
match = re.match(r'^[0-9]{4}_([0-9a-f]{16})\.[^.]+$', filename.lower())
|
||||
return match.group(1) if match else None
|
||||
|
||||
def get_file_hash(filepath: Path) -> str:
|
||||
"""Get hash for a file using dctfilename."""
|
||||
try:
|
||||
# Create a temporary file for processing
|
||||
with tempfile.NamedTemporaryFile(suffix=filepath.suffix) as tmp:
|
||||
temp_path = Path(tmp.name)
|
||||
try:
|
||||
# Try hardlink first
|
||||
os.link(str(filepath), tmp.name)
|
||||
except OSError:
|
||||
# Fall back to copy if hardlink fails
|
||||
shutil.copy2(str(filepath), tmp.name)
|
||||
|
||||
result = subprocess.run(['dctfilename', tmp.name],
|
||||
capture_output=True, text=True, check=True)
|
||||
return result.stdout.strip()
|
||||
except subprocess.CalledProcessError as e:
|
||||
print(f"Error getting hash for {filepath}: {e}")
|
||||
return ""
|
||||
except Exception as e:
|
||||
print(f"Unexpected error processing {filepath}: {e}")
|
||||
return ""
|
||||
|
||||
def process_directory(directory: Path):
|
||||
"""Process a single directory."""
|
||||
print(f"Processing directory: {directory}")
|
||||
|
||||
# Track files and their information
|
||||
seen_hashes: Dict[str, Path] = {} # hash -> filepath
|
||||
hash_sizes: Dict[str, int] = {} # hash -> filesize
|
||||
|
||||
# Get all image files in directory
|
||||
image_files = sorted(
|
||||
path for path in directory.iterdir()
|
||||
if path.is_file() and path.suffix.lower() in {'.jpg', '.jpeg', '.png'}
|
||||
)
|
||||
|
||||
# First pass: gather information and handle duplicates
|
||||
kept_files = [] # Files to keep and rename
|
||||
for filepath in image_files:
|
||||
print(f"Processing: {filepath.name}")
|
||||
|
||||
# Get hash (either from filename or calculate)
|
||||
if is_processed_file(filepath.name):
|
||||
file_hash = extract_hash(filepath.name)
|
||||
print(f"Using existing hash from filename: {file_hash}")
|
||||
else:
|
||||
file_hash = get_file_hash(filepath)
|
||||
print(f"Calculated new hash: {file_hash}")
|
||||
|
||||
if not file_hash:
|
||||
print(f"Skipping {filepath.name} due to hash error")
|
||||
continue
|
||||
|
||||
filesize = filepath.stat().st_size
|
||||
|
||||
if file_hash in seen_hashes:
|
||||
# Found duplicate
|
||||
if filesize > hash_sizes[file_hash]:
|
||||
print(f"Found larger duplicate: {filepath.name} replaces {seen_hashes[file_hash].name}")
|
||||
seen_hashes[file_hash].unlink() # Remove smaller file
|
||||
seen_hashes[file_hash] = filepath
|
||||
hash_sizes[file_hash] = filesize
|
||||
kept_files.append((filepath, file_hash))
|
||||
else:
|
||||
print(f"Removing smaller duplicate: {filepath.name}")
|
||||
filepath.unlink()
|
||||
else:
|
||||
seen_hashes[file_hash] = filepath
|
||||
hash_sizes[file_hash] = filesize
|
||||
kept_files.append((filepath, file_hash))
|
||||
|
||||
# Second pass: rename files
|
||||
for index, (filepath, file_hash) in enumerate(sorted(kept_files, key=lambda x: x[0].name), 1):
|
||||
new_name = filepath.parent / f"{index:04d}_{file_hash}{filepath.suffix.lower()}"
|
||||
|
||||
if filepath != new_name:
|
||||
print(f"Renaming: {filepath.name} -> {new_name.name}")
|
||||
filepath.rename(new_name)
|
||||
|
||||
print(f"Done! Processed {len(kept_files)} files in {directory}")
|
||||
|
||||
def main():
|
||||
if not check_dependencies():
|
||||
sys.exit(1)
|
||||
|
||||
# Get directories to process
|
||||
if len(sys.argv) == 1:
|
||||
directories = [Path('.')]
|
||||
else:
|
||||
recursive = sys.argv[1] in {'-r', '--recursive'}
|
||||
start_idx = 2 if recursive else 1
|
||||
|
||||
if recursive:
|
||||
directories = []
|
||||
for dir_arg in sys.argv[start_idx:]:
|
||||
path = Path(dir_arg)
|
||||
if path.is_dir():
|
||||
directories.extend(p for p in path.rglob('.') if p.is_dir())
|
||||
else:
|
||||
directories = [Path(dir_arg) for dir_arg in sys.argv[start_idx:] if Path(dir_arg).is_dir()]
|
||||
|
||||
# Process each directory
|
||||
for directory in directories:
|
||||
try:
|
||||
process_directory(directory)
|
||||
except Exception as e:
|
||||
print(f"Error processing directory {directory}: {e}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user