#!/bin/bash

# Script to find duplicate files in Photography and Videography directories
# This will help identify files that need to be deduplicated before backup

echo "=== DUPLICATE FILE FINDER ==="
echo "Scanning Photography and Videography directories..."
echo ""

# Create temporary files for processing
TEMP_DIR=$(mktemp -d)
SIZE_FILE="$TEMP_DIR/sizes.txt"
HASH_FILE="$TEMP_DIR/hashes.txt"
DUPLICATES_FILE="$TEMP_DIR/duplicates.txt"

# Function to find files by size first (faster than hashing everything)
find_by_size() {
    echo "Step 1: Finding files with identical sizes..."
    
    # Find all media files and get their sizes
    find Photography Videography -type f \( \
        -iname "*.jpg" -o -iname "*.jpeg" -o -iname "*.png" -o \
        -iname "*.cr2" -o -iname "*.raw" -o -iname "*.tiff" -o -iname "*.tif" -o \
        -iname "*.mp4" -o -iname "*.mov" -o -iname "*.avi" -o -iname "*.mkv" -o -iname "*.m4v" \
    \) -exec stat -f "%z %N" {} \; | sort -n > "$SIZE_FILE"
    
    # Find sizes that appear more than once
    awk '{print $1}' "$SIZE_FILE" | uniq -d > "$TEMP_DIR/duplicate_sizes.txt"
    
    echo "Found $(wc -l < "$TEMP_DIR/duplicate_sizes.txt") different file sizes with potential duplicates"
}

# Function to hash files with same sizes
hash_potential_duplicates() {
    echo "Step 2: Computing checksums for files with identical sizes..."
    
    while read -r size; do
        # Get all files with this size
        grep "^$size " "$SIZE_FILE" | while read -r filesize filepath; do
            # Compute MD5 hash for files with duplicate sizes
            if [ -f "$filepath" ]; then
                md5 -q "$filepath" 2>/dev/null && echo "$filepath"
            fi
        done | paste - - >> "$HASH_FILE" 2>/dev/null
    done < "$TEMP_DIR/duplicate_sizes.txt"
}

# Function to find actual duplicates by hash
find_hash_duplicates() {
    echo "Step 3: Identifying actual duplicates by hash..."
    
    # Sort by hash and find duplicates
    sort "$HASH_FILE" | awk '{
        if ($1 == prev_hash) {
            if (!printed_header) {
                print "DUPLICATE SET - Hash: " $1
                print "  " prev_file
                printed_header = 1
            }
            print "  " $2
        } else {
            printed_header = 0
            prev_hash = $1
            prev_file = $2
        }
    }' > "$DUPLICATES_FILE"
}

# Main execution
find_by_size
hash_potential_duplicates
find_hash_duplicates

echo ""
echo "=== RESULTS ==="
if [ -s "$DUPLICATES_FILE" ]; then
    echo "Found duplicate files:"
    cat "$DUPLICATES_FILE"
    echo ""
    echo "Duplicate report saved to: duplicates_report.txt"
    cp "$DUPLICATES_FILE" duplicates_report.txt
else
    echo "No duplicate files found!"
fi

# Cleanup
rm -rf "$TEMP_DIR"

echo ""
echo "=== SUMMARY BY YEAR ==="
echo "Photography files by year:"
find Photography -type f \( -iname "*.jpg" -o -iname "*.jpeg" -o -iname "*.png" -o -iname "*.cr2" -o -iname "*.raw" -o -iname "*.tiff" -o -iname "*.tif" \) | grep -E "(202[0-9])" | sed 's/.*(\([0-9]\{4\}\)).*/\1/' | sort | uniq -c

echo ""
echo "Videography files by year:"
find Videography -type f \( -iname "*.mp4" -o -iname "*.mov" -o -iname "*.avi" -o -iname "*.mkv" -o -iname "*.m4v" \) | grep -E "202[0-9]" | sed 's/.*\/\(202[0-9]\)\/.*/\1/' | sort | uniq -c