# SPDX-FileCopyrightText: 2026 Dennis Fink # # SPDX-License-Identifier: BSD-3-Clause """Render duplicate-file groups in terminal-friendly output formats.""" import shlex from json import dumps as json_dumps import click_extra as click import humanize from tabulate import tabulate from .types import FilesByHash OUTPUTS = {} def register_output(f): OUTPUTS[f.__name__] = f return f @register_output def human( duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False ) -> None: """Print duplicate groups in a human-readable format. :param duplicates: Hash digests mapped to files sharing each digest. :param show_size: Whether to display the size of each duplicate group. :param human_readable: Whether displayed sizes should use human-readable units. """ for hash_digest, files in duplicates.items(): click.secho("Duplicate set ", fg="green", bold=True, nl=False) click.secho("[", fg="magenta", bold=True, nl=False) click.secho(hash_digest, fg="cyan", bold=True, nl=False) click.secho("]", fg="magenta", bold=True, nl=False) if show_size: size = next(iter(files)).size click.secho(" [", fg="magenta", bold=True, nl=False) click.secho( humanize.naturalsize(size) if human_readable else size, fg="cyan", bold=True, nl=False, ) click.secho("]", fg="magenta", bold=True, nl=False) click.secho(":", fg="green", bold=True) for file in files: click.echo(f" {file.path}") @register_output def plain( duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False ) -> None: """Print duplicate groups as shell-quoted, space-separated fields. Each duplicate group is written on a separate line. The hash digest is the first field, followed by the duplicate paths. If requested, the file size is appended as the final field. :param duplicates: Hash digests mapped to files sharing each digest. :param show_size: Whether to append the size of each duplicate group. :param human_readable: Whether displayed sizes should use human-readable units. """ for hash_digest, files in duplicates.items(): fields = [hash_digest] fields.extend(str(file.path) for file in files) if show_size: size = next(iter(files)).size fields.append(humanize.naturalsize(size) if human_readable else str(size)) click.echo(shlex.join(fields)) @register_output def table( duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False ) -> None: """Print duplicate groups as a column-aligned table. :param duplicates: Hash digests mapped to files sharing each digest. :param show_size: Whether to display the size of each duplicate group. :param human_readable: Whether displayed sizes should use human-readable units. """ rows = [] for hash_digest, files in duplicates.items(): for file in files: if show_size: rows.append( ( click.style(hash_digest, fg="cyan"), file.path, humanize.naturalsize(file.size) if human_readable else file.size, ) ) else: rows.append((click.style(hash_digest, fg="cyan"), file.path)) click.echo(tabulate(rows, tablefmt="plain")) @register_output def json( duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False ) -> None: """Print duplicate groups as JSON. File paths are serialized as absolute strings. ``show_size`` is accepted for the common output-function interface but does not change the JSON structure. :param duplicates: Hash digests mapped to files sharing each digest. :param show_size: Unused; accepted for consistency with other output formats. :param human_readable: Whether the JSON should be pretty-printed. """ json_config = {"separators": (",", ":"), "indent": 0, "sort_keys": False} if human_readable: json_config["separators"] = (",", ": ") json_config["indent"] = 2 json_config["sort_keys"] = True serialized_duplicates = { hash_digest: [str(file.path.absolute()) for file in files] for hash_digest, files in duplicates.items() } click.echo(json_dumps(serialized_duplicates, **json_config))