1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
|
# SPDX-FileCopyrightText: 2026 Dennis Fink <me+coding@dennisfink.me>
#
# SPDX-License-Identifier: BSD-3-Clause
"""Render duplicate-file groups in terminal-friendly output formats."""
import shlex
from json import dumps as json_dumps
import click_extra as click
import humanize
from tabulate import tabulate
from .types import FilesByHash
OUTPUTS = {}
def register_output(f):
OUTPUTS[f.__name__] = f
return f
@register_output
def human(
duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False
) -> None:
"""Print duplicate groups in a human-readable format.
:param duplicates: Hash digests mapped to files sharing each digest.
:param show_size: Whether to display the size of each duplicate group.
:param human_readable: Whether displayed sizes should use human-readable units.
"""
for hash_digest, files in duplicates.items():
click.secho("Duplicate set ", fg="green", bold=True, nl=False)
click.secho("[", fg="magenta", bold=True, nl=False)
click.secho(hash_digest, fg="cyan", bold=True, nl=False)
click.secho("]", fg="magenta", bold=True, nl=False)
if show_size:
size = next(iter(files)).size
click.secho(" [", fg="magenta", bold=True, nl=False)
click.secho(
humanize.naturalsize(size) if human_readable else size,
fg="cyan",
bold=True,
nl=False,
)
click.secho("]", fg="magenta", bold=True, nl=False)
click.secho(":", fg="green", bold=True)
for file in files:
click.echo(f" {file.path}")
@register_output
def plain(
duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False
) -> None:
"""Print duplicate groups as shell-quoted, space-separated fields.
Each duplicate group is written on a separate line. The hash digest is the
first field, followed by the duplicate paths. If requested, the file size is
appended as the final field.
:param duplicates: Hash digests mapped to files sharing each digest.
:param show_size: Whether to append the size of each duplicate group.
:param human_readable: Whether displayed sizes should use human-readable units.
"""
for hash_digest, files in duplicates.items():
fields = [hash_digest]
fields.extend(str(file.path) for file in files)
if show_size:
size = next(iter(files)).size
fields.append(humanize.naturalsize(size) if human_readable else str(size))
click.echo(shlex.join(fields))
@register_output
def table(
duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False
) -> None:
"""Print duplicate groups as a column-aligned table.
:param duplicates: Hash digests mapped to files sharing each digest.
:param show_size: Whether to display the size of each duplicate group.
:param human_readable: Whether displayed sizes should use human-readable units.
"""
rows = []
for hash_digest, files in duplicates.items():
for file in files:
if show_size:
rows.append(
(
click.style(hash_digest, fg="cyan"),
file.path,
humanize.naturalsize(file.size)
if human_readable
else file.size,
)
)
else:
rows.append((click.style(hash_digest, fg="cyan"), file.path))
click.echo(tabulate(rows, tablefmt="plain"))
@register_output
def json(
duplicates: FilesByHash, show_size: bool = False, human_readable: bool = False
) -> None:
"""Print duplicate groups as JSON.
File paths are serialized as absolute strings. ``show_size`` is accepted for
the common output-function interface but does not change the JSON structure.
:param duplicates: Hash digests mapped to files sharing each digest.
:param show_size: Unused; accepted for consistency with other output formats.
:param human_readable: Whether the JSON should be pretty-printed.
"""
json_config = {"separators": (",", ":"), "indent": 0, "sort_keys": False}
if human_readable:
json_config["separators"] = (",", ": ")
json_config["indent"] = 2
json_config["sort_keys"] = True
serialized_duplicates = {
hash_digest: [str(file.path.absolute()) for file in files]
for hash_digest, files in duplicates.items()
}
click.echo(json_dumps(serialized_duplicates, **json_config))
|