aboutsummaryrefslogtreecommitdiff
path: root/duplicate_finder/__init__.py
diff options
context:
space:
mode:
Diffstat (limited to 'duplicate_finder/__init__.py')
-rw-r--r--duplicate_finder/__init__.py330
1 files changed, 330 insertions, 0 deletions
diff --git a/duplicate_finder/__init__.py b/duplicate_finder/__init__.py
new file mode 100644
index 0000000..b99c0ca
--- /dev/null
+++ b/duplicate_finder/__init__.py
@@ -0,0 +1,330 @@
+# SPDX-FileCopyrightText: 2026 Dennis Fink <me+coding@dennisfink.me>
+#
+# SPDX-License-Identifier: BSD-3-Clause
+
+###############################################################################
+# DESCRIPTION
+# This script scans one or more paths and identifies duplicate files.
+# Candidates are grouped by size and file extension before their contents
+# are hashed. The output can be rendered as human-readable text, plain text,
+# a table, or structured JSON.
+#
+# NOTES
+# This script respects the NO_COLOR standard (https://no-color.org/).
+#
+# AUTHOR
+# Dennis Fink <me+coding@dennisfink.me>
+###############################################################################
+
+"""Provide the command-line interface for finding duplicate files.
+
+Files are discovered recursively and grouped into candidate sets by size and,
+by default, file extension. Paths in each candidate set are also grouped by
+filesystem identity so hard links can be handled efficiently. Candidate files
+are then hashed to identify matching content.
+"""
+
+import hashlib
+import os
+from pathlib import Path
+from typing import TextIO
+
+import click_extra as click
+
+from . import output
+from .cli import ByteSizeParamType, FlexibleColorOption, debug, msg
+from .hash import group_files_by_hash
+from .scanner import group_file_candidates, iter_files_from_paths
+
+VERSION = "1.0.0"
+DESCRIPTION = "Find duplicate files using metadata grouping and content hashes."
+DATE_OF_CREATION = "2017-04-23"
+DATE_OF_REVISION = "2026-09-20"
+AUTHOR = "Dennis Fink <me+coding@dennisfink.me>"
+LICENSE = "BSD-3-Clause"
+
+
+def print_version(
+ ctx: click.Context, param: click.Parameter | None, value: bool
+) -> None:
+ """Print script metadata and exit.
+
+ :param ctx: Click context associated with the currently running command.
+ :param param: Click parameter that triggered the callback, if available.
+ :param value: Whether the version option was supplied.
+ """
+ if not value or ctx.resilient_parsing:
+ return
+
+ click.echo(click.style("Scriptname:", fg="red", bold=True) + f" {ctx.info_name}")
+ click.echo(click.style("Version:", fg="green", bold=True) + f" {VERSION}")
+ click.echo(click.style("Description:", fg="yellow", bold=True) + f" {DESCRIPTION}")
+ click.echo(click.style("Author:", fg="blue", bold=True) + f" {AUTHOR}")
+ click.echo(
+ click.style("Date of creation:", fg="magenta", bold=True)
+ + f" {DATE_OF_CREATION}"
+ )
+ click.echo(
+ click.style("Date of revision:", fg="cyan", bold=True) + f" {DATE_OF_REVISION}"
+ )
+ click.echo(click.style("License:", fg="red", bold=True) + f" {LICENSE}")
+
+ click.echo("""Copyright (c) 2026 Dennis Fink <me+coding@dennisfink.me>.
+
+Redistribution and use in source and binary forms, with or without
+modification, are permitted provided that the following conditions are met:
+
+1. Redistributions of source code must retain the above copyright notice, this
+list of conditions and the following disclaimer.
+
+2. Redistributions in binary form must reproduce the above copyright notice,
+this list of conditions and the following disclaimer in the documentation
+and/or other materials provided with the distribution.
+
+3. Neither the name of the copyright holder nor the names of its contributors
+may be used to endorse or promote products derived from this software without
+specific prior written permission.
+
+THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS \"AS IS\" AND
+ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
+WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
+DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
+FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
+DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
+SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
+CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
+OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
+OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.""")
+
+ ctx.exit()
+
+
+@click.command(
+ context_settings={"help_option_names": ("-h", "--help", "-?")}, params=[]
+)
+@click.option(
+ "--read-from",
+ type=click.File("r", encoding="utf-8"),
+ metavar="FILE",
+ help=(
+ "Read files and directories to scan from FILE, one path per line. "
+ "Use '-' for standard input. When set, positional paths are ignored."
+ ),
+)
+@click.option(
+ "--follow-symlinks/--no-follow-symlinks",
+ "follow_symlinks_flag",
+ default=False,
+ help="Follow symlinks when scanning directories.",
+)
+@click.option(
+ "-H",
+ "--include-hardlinks/--exclude-hardlinks",
+ "hardlinks_flag",
+ default=False,
+ help=(
+ "Include hard links in duplicate results, so multiple paths to the same "
+ "underlying file may be listed as duplicates."
+ ),
+)
+@click.option(
+ "-G",
+ "--minsize",
+ type=ByteSizeParamType(),
+ help="Consider only files >= SIZE bytes (supports suffixes like 10M, 1G, 500K).",
+)
+@click.option(
+ "-L",
+ "--maxsize",
+ type=ByteSizeParamType(),
+ help="Consider only files <= SIZE bytes (supports suffixes like 10M, 1G, 500K).",
+)
+@click.option(
+ "--include-empty/--exclude-empty",
+ "include_empty_flag",
+ default=False,
+ help="Include files that are 0 bytes in size.",
+)
+@click.option(
+ "--include-file-extension/--exclude-file-extension",
+ "include_file_extension_flag",
+ default=True,
+ help=(
+ "Include file extensions when grouping duplicate candidates. "
+ "Enabled by default; excluding extensions groups candidates by size only."
+ ),
+)
+@click.option(
+ "--hash",
+ "hash_func_name",
+ default="xxh128",
+ help="Select which hash function to use. Default xxh128.",
+ type=click.Choice(
+ sorted(["xxh32", "xxh64", "xxh128"] + list(hashlib.algorithms_available))
+ ),
+)
+@click.option(
+ "-j",
+ "--jobs",
+ default=0,
+ help="Maximum number of worker threads (0 = use default).",
+ type=click.IntRange(min=0),
+)
+@click.option(
+ "-o",
+ "--output-format",
+ default="human",
+ help=(
+ "Select the output format for duplicate groups. "
+ "'human' shows a readable text list, "
+ "'plain' prints each duplicate group as a space-separated line, "
+ "'table' prints a column-aligned overview, "
+ "and 'json' outputs machine-readable structured data."
+ ),
+ type=click.Choice(output.OUTPUTS.keys(), case_sensitive=False),
+)
+@click.option(
+ "-S",
+ "--size",
+ "show_size_flag",
+ is_flag=True,
+ default=False,
+ help="Show size of duplicate files (human/plain/table output only).",
+)
+@click.option(
+ "--human-readable/--no-human-readable", "human_readable_flag", default=False
+)
+@click.option(
+ "-q",
+ "--quiet",
+ "quiet_flag",
+ is_flag=True,
+ default=False,
+ help="Suppress all non-error output.",
+)
+@click.option(
+ "-v",
+ "--verbose",
+ "verbose_flag",
+ is_flag=True,
+ default=False,
+ help="Enable verbose output.",
+)
+@click.option("--color", cls=FlexibleColorOption)
+@click.no_color_option()
+@click.option(
+ "--version",
+ callback=print_version,
+ default=False,
+ expose_value=False,
+ help="Show the version and exit.",
+ is_eager=True,
+ is_flag=True,
+)
+@click.argument(
+ "paths",
+ type=click.Path(
+ exists=True, file_okay=True, dir_okay=True, readable=True, path_type=Path
+ ),
+ nargs=-1,
+)
+@click.pass_context
+def duplicate_finder(
+ ctx: click.Context,
+ paths: tuple[Path] | None = None,
+ read_from: TextIO | None = None,
+ follow_symlinks_flag: bool = False,
+ hardlinks_flag: bool = False,
+ minsize: int | None = None,
+ maxsize: int | None = None,
+ include_empty_flag: bool = False,
+ include_file_extension_flag: bool = True,
+ hash_func_name: str = "xxh128",
+ jobs: int = 0,
+ output_format: str = "human",
+ show_size_flag: bool = False,
+ human_readable_flag: bool = False,
+ quiet_flag: bool = False,
+ verbose_flag: bool = False,
+) -> None:
+ """Scan paths and list duplicate files.
+
+ Files are first grouped into candidate sets by size and, by default,
+ case-insensitive file extension. ``--exclude-file-extension`` groups
+ candidates by size only. Candidate files are then hashed to identify
+ matching content.
+ """
+ ctx.ensure_object(dict)
+ ctx.obj["debug"] = os.environ.get("DEBUG", "").lower() in ("1", "true", "yes")
+ ctx.obj["quiet"] = quiet_flag
+ ctx.obj["verbose"] = verbose_flag
+
+ debug("Effective configuration:", err=True)
+ debug(" paths:", ", ".join(str(p) for p in paths) if paths else ".", err=True)
+ debug(
+ " read_from:",
+ str(read_from.name) if read_from is not None else str(read_from),
+ err=True,
+ )
+ debug(" follow_symlinks:", str(follow_symlinks_flag), err=True)
+ debug(" hardlinks:", str(hardlinks_flag), err=True)
+ debug(" minsize:", str(minsize), err=True)
+ debug(" maxsize:", str(maxsize), err=True)
+ debug(" include_empty:", str(include_empty_flag), err=True)
+ debug(" include_file_extension:", str(include_file_extension_flag), err=True)
+ debug(" hash:", str(hash_func_name), err=True)
+ debug(" jobs:", "default" if jobs == 0 else str(jobs), err=True)
+ debug(" output_format:", str(output_format), err=True)
+ debug(" size:", str(show_size_flag), err=True)
+ debug(" human_readable:", str(human_readable_flag), err=True)
+ debug(" quiet:", str(quiet_flag), err=True)
+ debug(" verbose:", str(verbose_flag), err=True)
+
+ if read_from is not None:
+ input_paths = (
+ Path(path) for line in read_from if (path := line.rstrip("\r\n"))
+ )
+ else:
+ input_paths = iter(paths or (Path("."),))
+
+ candidate_files = group_file_candidates(
+ iter_files_from_paths(input_paths, follow_symlinks=follow_symlinks_flag),
+ minsize=minsize,
+ maxsize=maxsize,
+ include_empty=include_empty_flag,
+ include_file_extension=include_file_extension_flag,
+ include_hardlinks=hardlinks_flag,
+ )
+ debug(
+ "Candidate grouping complete:",
+ f"{len(candidate_files)} candidate groups remain for hashing.",
+ err=True,
+ )
+
+ hash_groups = group_files_by_hash(
+ candidate_files,
+ include_hardlinks=hardlinks_flag,
+ hash_name=hash_func_name,
+ jobs=jobs,
+ )
+
+ if ctx.obj["debug"]:
+ duplicate_paths = sum(len(files) for files in hash_groups.values())
+ debug(
+ "Duplicate detection complete:",
+ f"{len(hash_groups)} groups containing {duplicate_paths} paths.",
+ err=True,
+ )
+
+ if not hash_groups and output_format != "json":
+ msg("No duplicates found!", err=True)
+ ctx.exit(2)
+
+ debug(f"Rendering results using {output_format!r} output.", err=True)
+ output.OUTPUTS[output_format](hash_groups, show_size_flag, human_readable_flag)
+
+ ctx.exit()
+
+
+if __name__ == "__main__":
+ duplicate_finder()