aboutsummaryrefslogtreecommitdiff
path: root/duplicate_finder/output.py
diff options
context:
space:
mode:
authorDennis Fink2026-09-20 20:04:13 +0200
committerDennis Fink2026-09-20 20:04:13 +0200
commitb751174b4ed7c23786ae54a1bbaf332f653faa81 (patch)
treec1951961b7bdd171aa4903d9e2d5ad4f2a063b51 /duplicate_finder/output.py
downloadduplicate-finder-1.0.0.tar.gz
duplicate-finder-1.0.0.zip
feat: release duplicate-finder 1.0.0HEADv1.0.0main
Introduce the first public release of duplicate-finder, a command-line tool for detecting duplicate files through metadata grouping and content hashing. Support recursive path scanning, optional symlink traversal, hard-link handling, size and extension-based candidate grouping, and parallel content hashing with xxHash or hashlib algorithms. Provide human, plain, table, and JSON output formats together with size filtering, path input from files or stdin, progress reporting, diagnostic output, and configurable terminal colors. Add Python packaging for Python 3.14, project documentation, dependency locking, development tooling, and BSD-3-Clause licensing.
Diffstat (limited to '')
-rw-r--r--duplicate_finder/output.py136
1 files changed, 136 insertions, 0 deletions
diff --git a/duplicate_finder/output.py b/duplicate_finder/output.py
new file mode 100644
index 0000000..c3b6307
--- /dev/null
+++ b/duplicate_finder/output.py
@@ -0,0 +1,136 @@
+# SPDX-FileCopyrightText: 2026 Dennis Fink <me+coding@dennisfink.me>
+#
+# SPDX-License-Identifier: BSD-3-Clause
+
+"""Render duplicate-file groups in terminal-friendly output formats."""
+
+import shlex
+from json import dumps as json_dumps
+
+import click_extra as click
+import humanize
+from tabulate import tabulate
+
+from .types import FilesByHash
+
+OUTPUTS = {}
+
+
+def register_output(f):
+ OUTPUTS[f.__name__] = f
+ return f
+
+
+@register_output
+def human(
+ duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False
+) -> None:
+ """Print duplicate groups in a human-readable format.
+
+ :param duplicates: Hash digests mapped to files sharing each digest.
+ :param show_size: Whether to display the size of each duplicate group.
+ :param human_readable: Whether displayed sizes should use human-readable units.
+ """
+ for hash_digest, files in duplicates.items():
+ click.secho("Duplicate set ", fg="green", bold=True, nl=False)
+ click.secho("[", fg="magenta", bold=True, nl=False)
+ click.secho(hash_digest, fg="cyan", bold=True, nl=False)
+ click.secho("]", fg="magenta", bold=True, nl=False)
+
+ if show_size:
+ size = next(iter(files)).size
+ click.secho(" [", fg="magenta", bold=True, nl=False)
+ click.secho(
+ humanize.naturalsize(size) if human_readable else size,
+ fg="cyan",
+ bold=True,
+ nl=False,
+ )
+ click.secho("]", fg="magenta", bold=True, nl=False)
+
+ click.secho(":", fg="green", bold=True)
+
+ for file in files:
+ click.echo(f" {file.path}")
+
+
+@register_output
+def plain(
+ duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False
+) -> None:
+ """Print duplicate groups as shell-quoted, space-separated fields.
+
+ Each duplicate group is written on a separate line. The hash digest is the
+ first field, followed by the duplicate paths. If requested, the file size is
+ appended as the final field.
+
+ :param duplicates: Hash digests mapped to files sharing each digest.
+ :param show_size: Whether to append the size of each duplicate group.
+ :param human_readable: Whether displayed sizes should use human-readable units.
+ """
+ for hash_digest, files in duplicates.items():
+ fields = [hash_digest]
+ fields.extend(str(file.path) for file in files)
+
+ if show_size:
+ size = next(iter(files)).size
+ fields.append(humanize.naturalsize(size) if human_readable else str(size))
+
+ click.echo(shlex.join(fields))
+
+
+@register_output
+def table(
+ duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False
+) -> None:
+ """Print duplicate groups as a column-aligned table.
+
+ :param duplicates: Hash digests mapped to files sharing each digest.
+ :param show_size: Whether to display the size of each duplicate group.
+ :param human_readable: Whether displayed sizes should use human-readable units.
+ """
+ rows = []
+ for hash_digest, files in duplicates.items():
+ for file in files:
+ if show_size:
+ rows.append(
+ (
+ click.style(hash_digest, fg="cyan"),
+ file.path,
+ humanize.naturalsize(file.size)
+ if human_readable
+ else file.size,
+ )
+ )
+ else:
+ rows.append((click.style(hash_digest, fg="cyan"), file.path))
+
+ click.echo(tabulate(rows, tablefmt="plain"))
+
+
+@register_output
+def json(
+ duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False
+) -> None:
+ """Print duplicate groups as JSON.
+
+ File paths are serialized as absolute strings. ``show_size`` is accepted for
+ the common output-function interface but does not change the JSON structure.
+
+ :param duplicates: Hash digests mapped to files sharing each digest.
+ :param show_size: Unused; accepted for consistency with other output formats.
+ :param human_readable: Whether the JSON should be pretty-printed.
+ """
+ json_config = {"separators": (",", ":"), "indent": 0, "sort_keys": False}
+
+ if human_readable:
+ json_config["separators"] = (",", ": ")
+ json_config["indent"] = 2
+ json_config["sort_keys"] = True
+
+ serialized_duplicates = {
+ hash_digest: [str(file.path.absolute()) for file in files]
+ for hash_digest, files in duplicates.items()
+ }
+
+ click.echo(json_dumps(serialized_duplicates, **json_config))