diff options
Diffstat (limited to '')
| -rw-r--r-- | duplicate_finder/__init__.py | 330 |
1 files changed, 330 insertions, 0 deletions
diff --git a/duplicate_finder/__init__.py b/duplicate_finder/__init__.py new file mode 100644 index 0000000..b99c0ca --- /dev/null +++ b/duplicate_finder/__init__.py @@ -0,0 +1,330 @@ +# SPDX-FileCopyrightText: 2026 Dennis Fink <me+coding@dennisfink.me> +# +# SPDX-License-Identifier: BSD-3-Clause + +############################################################################### +# DESCRIPTION +# This script scans one or more paths and identifies duplicate files. +# Candidates are grouped by size and file extension before their contents +# are hashed. The output can be rendered as human-readable text, plain text, +# a table, or structured JSON. +# +# NOTES +# This script respects the NO_COLOR standard (https://no-color.org/). +# +# AUTHOR +# Dennis Fink <me+coding@dennisfink.me> +############################################################################### + +"""Provide the command-line interface for finding duplicate files. + +Files are discovered recursively and grouped into candidate sets by size and, +by default, file extension. Paths in each candidate set are also grouped by +filesystem identity so hard links can be handled efficiently. Candidate files +are then hashed to identify matching content. +""" + +import hashlib +import os +from pathlib import Path +from typing import TextIO + +import click_extra as click + +from . import output +from .cli import ByteSizeParamType, FlexibleColorOption, debug, msg +from .hash import group_files_by_hash +from .scanner import group_file_candidates, iter_files_from_paths + +VERSION = "1.0.0" +DESCRIPTION = "Find duplicate files using metadata grouping and content hashes." +DATE_OF_CREATION = "2017-04-23" +DATE_OF_REVISION = "2026-09-20" +AUTHOR = "Dennis Fink <me+coding@dennisfink.me>" +LICENSE = "BSD-3-Clause" + + +def print_version( + ctx: click.Context, param: click.Parameter | None, value: bool +) -> None: + """Print script metadata and exit. + + :param ctx: Click context associated with the currently running command. + :param param: Click parameter that triggered the callback, if available. + :param value: Whether the version option was supplied. + """ + if not value or ctx.resilient_parsing: + return + + click.echo(click.style("Scriptname:", fg="red", bold=True) + f" {ctx.info_name}") + click.echo(click.style("Version:", fg="green", bold=True) + f" {VERSION}") + click.echo(click.style("Description:", fg="yellow", bold=True) + f" {DESCRIPTION}") + click.echo(click.style("Author:", fg="blue", bold=True) + f" {AUTHOR}") + click.echo( + click.style("Date of creation:", fg="magenta", bold=True) + + f" {DATE_OF_CREATION}" + ) + click.echo( + click.style("Date of revision:", fg="cyan", bold=True) + f" {DATE_OF_REVISION}" + ) + click.echo(click.style("License:", fg="red", bold=True) + f" {LICENSE}") + + click.echo("""Copyright (c) 2026 Dennis Fink <me+coding@dennisfink.me>. + +Redistribution and use in source and binary forms, with or without +modification, are permitted provided that the following conditions are met: + +1. Redistributions of source code must retain the above copyright notice, this +list of conditions and the following disclaimer. + +2. Redistributions in binary form must reproduce the above copyright notice, +this list of conditions and the following disclaimer in the documentation +and/or other materials provided with the distribution. + +3. Neither the name of the copyright holder nor the names of its contributors +may be used to endorse or promote products derived from this software without +specific prior written permission. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS \"AS IS\" AND +ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED +WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE +DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE +FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL +DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR +SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER +CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, +OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE +OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.""") + + ctx.exit() + + +@click.command( + context_settings={"help_option_names": ("-h", "--help", "-?")}, params=[] +) +@click.option( + "--read-from", + type=click.File("r", encoding="utf-8"), + metavar="FILE", + help=( + "Read files and directories to scan from FILE, one path per line. " + "Use '-' for standard input. When set, positional paths are ignored." + ), +) +@click.option( + "--follow-symlinks/--no-follow-symlinks", + "follow_symlinks_flag", + default=False, + help="Follow symlinks when scanning directories.", +) +@click.option( + "-H", + "--include-hardlinks/--exclude-hardlinks", + "hardlinks_flag", + default=False, + help=( + "Include hard links in duplicate results, so multiple paths to the same " + "underlying file may be listed as duplicates." + ), +) +@click.option( + "-G", + "--minsize", + type=ByteSizeParamType(), + help="Consider only files >= SIZE bytes (supports suffixes like 10M, 1G, 500K).", +) +@click.option( + "-L", + "--maxsize", + type=ByteSizeParamType(), + help="Consider only files <= SIZE bytes (supports suffixes like 10M, 1G, 500K).", +) +@click.option( + "--include-empty/--exclude-empty", + "include_empty_flag", + default=False, + help="Include files that are 0 bytes in size.", +) +@click.option( + "--include-file-extension/--exclude-file-extension", + "include_file_extension_flag", + default=True, + help=( + "Include file extensions when grouping duplicate candidates. " + "Enabled by default; excluding extensions groups candidates by size only." + ), +) +@click.option( + "--hash", + "hash_func_name", + default="xxh128", + help="Select which hash function to use. Default xxh128.", + type=click.Choice( + sorted(["xxh32", "xxh64", "xxh128"] + list(hashlib.algorithms_available)) + ), +) +@click.option( + "-j", + "--jobs", + default=0, + help="Maximum number of worker threads (0 = use default).", + type=click.IntRange(min=0), +) +@click.option( + "-o", + "--output-format", + default="human", + help=( + "Select the output format for duplicate groups. " + "'human' shows a readable text list, " + "'plain' prints each duplicate group as a space-separated line, " + "'table' prints a column-aligned overview, " + "and 'json' outputs machine-readable structured data." + ), + type=click.Choice(output.OUTPUTS.keys(), case_sensitive=False), +) +@click.option( + "-S", + "--size", + "show_size_flag", + is_flag=True, + default=False, + help="Show size of duplicate files (human/plain/table output only).", +) +@click.option( + "--human-readable/--no-human-readable", "human_readable_flag", default=False +) +@click.option( + "-q", + "--quiet", + "quiet_flag", + is_flag=True, + default=False, + help="Suppress all non-error output.", +) +@click.option( + "-v", + "--verbose", + "verbose_flag", + is_flag=True, + default=False, + help="Enable verbose output.", +) +@click.option("--color", cls=FlexibleColorOption) +@click.no_color_option() +@click.option( + "--version", + callback=print_version, + default=False, + expose_value=False, + help="Show the version and exit.", + is_eager=True, + is_flag=True, +) +@click.argument( + "paths", + type=click.Path( + exists=True, file_okay=True, dir_okay=True, readable=True, path_type=Path + ), + nargs=-1, +) +@click.pass_context +def duplicate_finder( + ctx: click.Context, + paths: tuple[Path] | None = None, + read_from: TextIO | None = None, + follow_symlinks_flag: bool = False, + hardlinks_flag: bool = False, + minsize: int | None = None, + maxsize: int | None = None, + include_empty_flag: bool = False, + include_file_extension_flag: bool = True, + hash_func_name: str = "xxh128", + jobs: int = 0, + output_format: str = "human", + show_size_flag: bool = False, + human_readable_flag: bool = False, + quiet_flag: bool = False, + verbose_flag: bool = False, +) -> None: + """Scan paths and list duplicate files. + + Files are first grouped into candidate sets by size and, by default, + case-insensitive file extension. ``--exclude-file-extension`` groups + candidates by size only. Candidate files are then hashed to identify + matching content. + """ + ctx.ensure_object(dict) + ctx.obj["debug"] = os.environ.get("DEBUG", "").lower() in ("1", "true", "yes") + ctx.obj["quiet"] = quiet_flag + ctx.obj["verbose"] = verbose_flag + + debug("Effective configuration:", err=True) + debug(" paths:", ", ".join(str(p) for p in paths) if paths else ".", err=True) + debug( + " read_from:", + str(read_from.name) if read_from is not None else str(read_from), + err=True, + ) + debug(" follow_symlinks:", str(follow_symlinks_flag), err=True) + debug(" hardlinks:", str(hardlinks_flag), err=True) + debug(" minsize:", str(minsize), err=True) + debug(" maxsize:", str(maxsize), err=True) + debug(" include_empty:", str(include_empty_flag), err=True) + debug(" include_file_extension:", str(include_file_extension_flag), err=True) + debug(" hash:", str(hash_func_name), err=True) + debug(" jobs:", "default" if jobs == 0 else str(jobs), err=True) + debug(" output_format:", str(output_format), err=True) + debug(" size:", str(show_size_flag), err=True) + debug(" human_readable:", str(human_readable_flag), err=True) + debug(" quiet:", str(quiet_flag), err=True) + debug(" verbose:", str(verbose_flag), err=True) + + if read_from is not None: + input_paths = ( + Path(path) for line in read_from if (path := line.rstrip("\r\n")) + ) + else: + input_paths = iter(paths or (Path("."),)) + + candidate_files = group_file_candidates( + iter_files_from_paths(input_paths, follow_symlinks=follow_symlinks_flag), + minsize=minsize, + maxsize=maxsize, + include_empty=include_empty_flag, + include_file_extension=include_file_extension_flag, + include_hardlinks=hardlinks_flag, + ) + debug( + "Candidate grouping complete:", + f"{len(candidate_files)} candidate groups remain for hashing.", + err=True, + ) + + hash_groups = group_files_by_hash( + candidate_files, + include_hardlinks=hardlinks_flag, + hash_name=hash_func_name, + jobs=jobs, + ) + + if ctx.obj["debug"]: + duplicate_paths = sum(len(files) for files in hash_groups.values()) + debug( + "Duplicate detection complete:", + f"{len(hash_groups)} groups containing {duplicate_paths} paths.", + err=True, + ) + + if not hash_groups and output_format != "json": + msg("No duplicates found!", err=True) + ctx.exit(2) + + debug(f"Rendering results using {output_format!r} output.", err=True) + output.OUTPUTS[output_format](hash_groups, show_size_flag, human_readable_flag) + + ctx.exit() + + +if __name__ == "__main__": + duplicate_finder() |
