diff --git a/FileDeduplicator.Test/VerbOutputTests.cs b/FileDeduplicator.Test/VerbOutputTests.cs
index c2f1d05..cc38f7e 100644
--- a/FileDeduplicator.Test/VerbOutputTests.cs
+++ b/FileDeduplicator.Test/VerbOutputTests.cs
@@ -206,6 +206,59 @@ public void StatsBreaksDuplicatesDownByEachCopysOwnExtension()
}
}
+ ///
+ /// Empty files are never deleted, so Stats does not count them as duplicates or wasted copies and
+ /// agrees with DryRun on how many files a run would remove (ktsu-dev/FileDeduplicator#172).
+ ///
+ [TestMethod]
+ public void StatsLeavesEmptyFilesOutOfTheDuplicates()
+ {
+ // Arrange -- the tree from the issue: four empty files and one 2-byte pair
+ using TempTree tree = new();
+ _ = tree.Write("a/__init__.py", string.Empty);
+ _ = tree.Write("b/__init__.py", string.Empty);
+ _ = tree.Write("c/__init__.py", string.Empty);
+ _ = tree.Write(".gitkeep", string.Empty);
+ _ = tree.Write("A.JPG", "xy");
+ _ = tree.Write("b/copy.jpg", "xy");
+
+ // Act
+ string output = ConsoleCapture.Normalize(ConsoleCapture.Run(new Stats { PathString = tree.Root.WeakString }));
+ string dryRun = ConsoleCapture.Normalize(ConsoleCapture.Run(new DryRun { PathString = tree.Root.WeakString }));
+
+ // Assert
+ Assert.Contains("Files to delete: 1", dryRun);
+ Assert.Contains("Duplicate files: 1", output);
+ Assert.Contains("Duplicate groups: 1", output);
+ Assert.Contains("Empty duplicates (never deleted): 3", output);
+ Assert.Contains("Duplicate files by extension:\n.jpg: 1 file(s)\n", output);
+ Assert.DoesNotContain(".py:", output);
+ Assert.DoesNotContain("0 B each", output, "An empty group was listed among the largest groups.");
+ }
+
+ ///
+ /// .jpg and .JPG copies are one kind of file and are counted under one extension.
+ ///
+ [TestMethod]
+ public void StatsCountsExtensionsThatDifferOnlyInCaseTogether()
+ {
+ // Arrange -- the keepers are the shorter names a.jpg, b.jpg and c.JPG
+ using TempTree tree = new();
+ _ = tree.Write("a.jpg", "first");
+ _ = tree.Write("a2.JPG", "first");
+ _ = tree.Write("b.jpg", "second");
+ _ = tree.Write("b2.jpg", "second");
+ _ = tree.Write("c.JPG", "third");
+ _ = tree.Write("c2.JPG", "third");
+
+ // Act
+ string output = ConsoleCapture.Normalize(ConsoleCapture.Run(new Stats { PathString = tree.Root.WeakString }));
+
+ // Assert
+ Assert.Contains("Duplicate files by extension:\n.jpg: 3 file(s)\n", output);
+ Assert.DoesNotContain(".JPG:", output);
+ }
+
///
/// A file that cannot be hashed is named by its full path, so the failing copy can be found
/// among others sharing its file name.
diff --git a/FileDeduplicator/Verbs/Stats.cs b/FileDeduplicator/Verbs/Stats.cs
index 45d1acb..90159bc 100644
--- a/FileDeduplicator/Verbs/Stats.cs
+++ b/FileDeduplicator/Verbs/Stats.cs
@@ -53,14 +53,20 @@ internal override void Run(Stats options)
// Step 3: Compute statistics
Dictionary> hashGroups = Deduplicator.GroupByHash(fileHashes);
- IReadOnlyList duplicates = Deduplicator.FindDuplicates(hashGroups);
+ IReadOnlyList allGroups = Deduplicator.FindDuplicates(hashGroups);
+
+ // Empty files group together but are never deleted (Deduplicator.IsDeletable), so counting
+ // them here reported dozens of __init__.py files as redundant copies that DryRun and
+ // Deduplicate would keep. They are reported on their own line instead.
+ DuplicateGroup[] duplicates = [.. allGroups.Where(Deduplicator.IsDeletable)];
+ int emptyCopies = allGroups.Where(g => !Deduplicator.IsDeletable(g)).Sum(g => g.Files.Count - 1);
// Every count below is taken over the files that hashed. HashFiles drops a file it cannot
// read, so counting against the scanned list would report each one as a duplicate of nothing,
// and sizing it would throw if it had vanished since the scan.
long totalSize = Deduplicator.TotalSize(fileHashes.Keys);
int uniqueFiles = hashGroups.Count;
- int duplicateFiles = fileHashes.Count - uniqueFiles;
+ int duplicateFiles = fileHashes.Count - uniqueFiles - emptyCopies;
int unreadableFiles = files.Count - fileHashes.Count;
Console.WriteLine("=== FileDeduplicator Statistics ===");
@@ -74,9 +80,13 @@ internal override void Run(Stats options)
Console.WriteLine($"Total size: {DuplicateReport.FormatBytes(totalSize)}");
Console.WriteLine($"Unique files: {uniqueFiles}");
Console.WriteLine($"Duplicate files: {duplicateFiles}");
- Console.WriteLine($"Duplicate groups: {duplicates.Count}");
+ Console.WriteLine($"Duplicate groups: {duplicates.Length}");
+ if (emptyCopies > 0)
+ {
+ Console.WriteLine($"Empty duplicates (never deleted): {emptyCopies}");
+ }
- if (duplicates.Count > 0)
+ if (duplicates.Length > 0)
{
long wastedSpace = duplicates.Sum(g => g.FileSize * (g.Files.Count - 1));
Console.WriteLine($"Wasted space: {DuplicateReport.FormatBytes(wastedSpace)}");
@@ -122,7 +132,8 @@ private static Dictionary CountRedundantCopiesByExtension(IReadOnly
AbsoluteFilePath keeper = Deduplicator.SelectFileToKeep(group.Files);
IEnumerable extensions = group.Files
.Where(f => f != keeper)
- .Select(f => System.IO.Path.GetExtension(f.WeakString))
+ // Camera and phone folders mix .JPG and .jpg, which are one kind of file.
+ .Select(f => System.IO.Path.GetExtension(f.WeakString).ToLowerInvariant())
.Select(ext => string.IsNullOrEmpty(ext) ? "(no extension)" : ext);
foreach (string ext in extensions)