#!/usr/bin/env bash # # find_duplicate_pictures.sh # # Finds all picture files on the given filesystems/directories and reports # duplicates using TWO independent checks: # 1. Same file NAME (catches renamed-but-identical-looking candidates) # 2. Same CONTENT (SHA-256 hash of the file bytes — catches same image # saved under different names) # # A name alone is not reliable: different pictures can share the same name # (e.g. IMG_0001.jpg from different cameras), so content hashing is used as # the authoritative duplicate check. Size is compared first as a cheap # pre-filter before hashing. # # Usage: # ./find_duplicate_pictures.sh [DIR ...] # ./find_duplicate_pictures.sh # defaults to "/" (whole system) # # Examples: # ./find_duplicate_pictures.sh / /mnt/c /mnt/d # ./find_duplicate_pictures.sh ~/Pictures # # Options (environment variables): # HASH_CMD=sha256sum # override hash tool (sha256sum, shasum -a 256, md5sum) # FOLLOW_LINKS=1 # follow symlinks while scanning (default: off) # REPORT=report.txt # also write the report to a file # set -u # --------------------------------------------------------------------------- # Config # --------------------------------------------------------------------------- # Picture extensions (case-insensitive match at scan time) EXTENSIONS="jpg jpeg png gif bmp tiff tif webp heic heif svg ico raw cr2 nef arw dng psd avif jfif" # Directories that should never be scanned on a full-system run PRUNE_DIRS="/proc /sys /dev /run /snap /tmp" # --------------------------------------------------------------------------- # Pick a hashing tool # --------------------------------------------------------------------------- pick_hash_cmd() { if [ -n "${HASH_CMD:-}" ]; then echo "$HASH_CMD" elif command -v sha256sum >/dev/null 2>&1; then echo "sha256sum" elif command -v shasum >/dev/null 2>&1; then echo "shasum -a 256" elif command -v md5sum >/dev/null 2>&1; then echo "md5sum" # weaker, but fine as a last resort else echo "ERROR: no hashing tool found (sha256sum, shasum, md5sum)." >&2 exit 1 fi } HASH=$(pick_hash_cmd) # --------------------------------------------------------------------------- # Build the find expression for extensions: -iname '*.jpg' -o -iname '*.png' ... # --------------------------------------------------------------------------- build_ext_expr() { local expr=() local first=1 for ext in $EXTENSIONS; do if [ "$first" -eq 1 ]; then expr+=( -iname "*.${ext}" ) first=0 else expr+=( -o -iname "*.${ext}" ) fi done printf '%s\n' "${expr[@]}" } # --------------------------------------------------------------------------- # Arguments / scan roots # --------------------------------------------------------------------------- if [ "$#" -gt 0 ]; then ROOTS=("$@") else ROOTS=("/") fi FIND_LINK_FLAG=() if [ "${FOLLOW_LINKS:-0}" = "1" ]; then FIND_LINK_FLAG=(-L) fi # Build prune expression for system pseudo-filesystems PRUNE_EXPR=() for d in $PRUNE_DIRS; do PRUNE_EXPR+=( -path "$d" -prune -o ) done # Extension expression mapfile -t EXT_EXPR < <(build_ext_expr) # --------------------------------------------------------------------------- # Temp files # --------------------------------------------------------------------------- WORKDIR=$(mktemp -d "${TMPDIR:-/tmp}/duppic.XXXXXX") || exit 1 trap 'rm -rf "$WORKDIR"' EXIT ALL_FILES="$WORKDIR/all.tsv" # size \t name \t path BY_NAME="$WORKDIR/by_name.txt" # duplicate name groups BY_SIZE="$WORKDIR/by_size.txt" # size-collision candidates HASHES="$WORKDIR/hashes.tsv" # hash \t size \t name \t path BY_HASH="$WORKDIR/by_hash.txt" # duplicate content groups : > "$ALL_FILES" : > "$HASHES" # --------------------------------------------------------------------------- # Step 1: Scan filesystems # --------------------------------------------------------------------------- echo "=== Step 1: scanning for picture files ===" for root in "${ROOTS[@]}"; do if [ ! -d "$root" ]; then echo " ! skipping '$root' (not a directory)" >&2 continue fi echo " -> $root" # NUL-safe scan; store: size basename full path find "${FIND_LINK_FLAG[@]}" "$root" \ \( "${PRUNE_EXPR[@]}" \( "${EXT_EXPR[@]}" \) -type f -print0 \) 2>/dev/null | while IFS= read -r -d '' f; do size=$(stat -c '%s' -- "$f" 2>/dev/null || stat -f '%z' -- "$f" 2>/dev/null) [ -n "$size" ] || continue base=${f##*/} printf '%s\t%s\t%s\n' "$size" "$base" "$f" done >> "$ALL_FILES" done TOTAL=$(wc -l < "$ALL_FILES" | tr -d ' ') echo " found $TOTAL picture file(s)" echo if [ "$TOTAL" -eq 0 ]; then echo "Nothing to compare. Done." exit 0 fi # --------------------------------------------------------------------------- # Step 2: Duplicates by NAME (case-insensitive) # --------------------------------------------------------------------------- echo "=== Step 2: duplicates by file name ===" awk -F'\t' '{ print tolower($2) }' "$ALL_FILES" | sort | uniq -d > "$WORKDIR/dupnames" if [ -s "$WORKDIR/dupnames" ]; then while IFS= read -r name; do echo "--- name: $name ---" awk -F'\t' -v n="$name" 'tolower($2) == n { printf " %10s bytes %s\n", $1, $3 }' "$ALL_FILES" echo done < "$WORKDIR/dupnames" | tee "$BY_NAME" NAME_GROUPS=$(wc -l < "$WORKDIR/dupnames" | tr -d ' ') echo " $NAME_GROUPS name group(s) share the same file name" else echo " no duplicate names found" fi echo # --------------------------------------------------------------------------- # Step 3: Duplicates by CONTENT (size pre-filter + hash) # --------------------------------------------------------------------------- echo "=== Step 3: duplicates by content (SHA-256) ===" # Only files whose size appears more than once can possibly collide — # hashing only those saves a lot of time on large collections. awk -F'\t' '{ print $1 }' "$ALL_FILES" | sort | uniq -d > "$WORKDIR/dupsizes" if [ ! -s "$WORKDIR/dupsizes" ]; then echo " no files share the same size — content duplicates impossible" else CANDIDATES=$(awk -F'\t' 'NR==FNR { s[$1]=1; next } s[$1]' \ "$WORKDIR/dupsizes" "$ALL_FILES" | wc -l | tr -d ' ') echo " hashing $CANDIDATES candidate file(s) with matching sizes..." # Hash each candidate (skip unreadable files) while IFS=$'\t' read -r size base path; do h=$($HASH -- "$path" 2>/dev/null | awk '{print $1}') [ -n "$h" ] || continue printf '%s\t%s\t%s\t%s\n' "$h" "$size" "$base" "$path" done < <(awk -F'\t' 'NR==FNR { s[$1]=1; next } s[$1]' \ "$WORKDIR/dupsizes" "$ALL_FILES") > "$HASHES" # Groups of identical hashes = true duplicates cut -f1 "$HASHES" | sort | uniq -d > "$WORKDIR/duphashes" if [ -s "$WORKDIR/duphashes" ]; then while IFS= read -r h; do echo "--- identical content (sha256: ${h:0:16}...) ---" while IFS=$'\t' read -r gh gsize gname gpath; do [ "$gh" = "$h" ] || continue printf ' %10s bytes %s\n' "$gsize" "$gpath" done < "$HASHES" echo done < "$WORKDIR/duphashes" | tee "$BY_HASH" # space wasted by redundant copies = sum of size * (copies - 1) WASTED=$(awk -F'\t' '{ c[$1]++; s[$1]=$2 } END { w=0; for (h in c) if (c[h]>1) w += s[h]*(c[h]-1); print w }' \ "$HASHES") HASH_GROUPS=$(wc -l < "$WORKDIR/duphashes" | tr -d ' ') echo " $HASH_GROUPS content group(s) are true duplicates" echo " approx. space wasted by extra copies: $WASTED bytes" \ "($(numfmt --to=iec "$WASTED" 2>/dev/null || echo "$WASTED B"))" else echo " no identical content found (same-name files are different pictures)" fi fi echo # --------------------------------------------------------------------------- # Step 4: Cross-report — same NAME but DIFFERENT content # --------------------------------------------------------------------------- echo "=== Step 4: same name, different content (false friends) ===" if [ -s "$WORKDIR/dupnames" ]; then # hashed files contribute their hash, unhashed files (unique size, # so never hashed) are each distinct by definition. while IFS= read -r name; do distinct=$( { awk -F'\t' -v n="$name" 'tolower($3) == n { print "H:" $1 }' "$HASHES" 2>/dev/null awk -F'\t' -v n="$name" 'tolower($2) == n { print "P:" $3 }' "$ALL_FILES" | while IFS= read -r p; do p=${p#P:} grep -qF -- "$p" <(cut -f4 "$HASHES" 2>/dev/null) || printf 'U:%s\n' "$p" done } | sort -u | wc -l | tr -d ' ' ) if [ "$distinct" -gt 1 ]; then echo "--- '$name' is used by $distinct DIFFERENT pictures: ---" awk -F'\t' -v n="$name" 'tolower($3) == n { printf " [%s] %s\n", substr($1,1,12), $4 }' "$HASHES" 2>/dev/null # files that never got hashed (unique size) but share the name awk -F'\t' -v n="$name" 'tolower($2) == n' "$ALL_FILES" | while IFS=$'\t' read -r s b p; do grep -qF -- "$p" <(cut -f4 "$HASHES" 2>/dev/null) || \ printf ' [unique-size] %s\n' "$p" done echo fi done < "$WORKDIR/dupnames" | tee "$WORKDIR/false_friends.txt" grep -q '^--- ' "$WORKDIR/false_friends.txt" 2>/dev/null || \ echo " none — all same-named files are also identical in content" else echo " skipped (no duplicate names)" fi echo # --------------------------------------------------------------------------- # Optional report file # --------------------------------------------------------------------------- if [ -n "${REPORT:-}" ]; then { echo "Duplicate picture report — $(date)" echo "Roots: ${ROOTS[*]}" echo echo "### Duplicates by name ###" cat "$BY_NAME" 2>/dev/null || echo "none" echo "### Duplicates by content ###" cat "$BY_HASH" 2>/dev/null || echo "none" echo "### Same name, different content ###" cat "$WORKDIR/false_friends.txt" 2>/dev/null || echo "none" } > "$REPORT" echo "Report written to: $REPORT" fi echo "Done."