diff --git a/FileDeduplicator.Test/DeduplicatorTests.cs b/FileDeduplicator.Test/DeduplicatorTests.cs
index f792498..f1a6318 100644
--- a/FileDeduplicator.Test/DeduplicatorTests.cs
+++ b/FileDeduplicator.Test/DeduplicatorTests.cs
@@ -432,4 +432,73 @@ public void DeletingAnEmptyGroupListIsANoOp()
Assert.IsEmpty(result.Errors);
Assert.IsEmpty(result.SkippedFiles);
}
+
+ ///
+ /// A copy removed between the hash pass and grouping must not abort the run: the grouping
+ /// has to go on without it and still report the copies that remain.
+ ///
+ [TestMethod]
+ public void ACopyThatVanishedAfterHashingIsDroppedFromItsGroup()
+ {
+ // Arrange
+ using TempTree tree = new();
+ AbsoluteFilePath first = tree.Write("a.txt", "shared");
+ AbsoluteFilePath second = tree.Write("bb.txt", "shared");
+ AbsoluteFilePath vanishing = tree.Write("ccc.txt", "shared");
+ Dictionary hashes = FileHasher.HashFiles([first, second, vanishing]);
+
+ // Arrange -- something else removes it while the rest of a long hash pass runs
+ File.Delete(vanishing.WeakString);
+
+ // Act
+ IReadOnlyList duplicates = Duplicates(hashes);
+
+ // Assert
+ Assert.ContainsSingle(duplicates);
+ CollectionAssert.AreEquivalent(new[] { first, second }, duplicates[0].Files);
+ Assert.AreEqual("shared".Length, duplicates[0].FileSize);
+ }
+
+ ///
+ /// Whichever copy vanishes, including the one the group would have been sized from, a group
+ /// left with a single copy is no longer a duplicate group.
+ ///
+ /// Which of the two copies disappears.
+ [TestMethod]
+ [DataRow(0)]
+ [DataRow(1)]
+ public void AGroupLeftWithOneCopyIsDropped(int vanishingIndex)
+ {
+ // Arrange
+ using TempTree tree = new();
+ AbsoluteFilePath[] copies = [tree.Write("a.txt", "shared"), tree.Write("bb.txt", "shared")];
+ Dictionary hashes = FileHasher.HashFiles(copies);
+ File.Delete(copies[vanishingIndex].WeakString);
+
+ // Act
+ IReadOnlyList duplicates = Duplicates(hashes);
+
+ // Assert
+ Assert.IsEmpty(duplicates);
+ }
+
+ ///
+ /// Stats sizes every file that hashed; one that has since disappeared contributes nothing
+ /// rather than aborting the report.
+ ///
+ [TestMethod]
+ public void TotalSizeSkipsAFileThatVanishedAfterHashing()
+ {
+ // Arrange
+ using TempTree tree = new();
+ AbsoluteFilePath kept = tree.Write("a.txt", "four");
+ AbsoluteFilePath vanishing = tree.Write("c.txt", "unrelated");
+ File.Delete(vanishing.WeakString);
+
+ // Act
+ long total = Deduplicator.TotalSize([kept, vanishing]);
+
+ // Assert
+ Assert.AreEqual("four".Length, total);
+ }
}
diff --git a/FileDeduplicator/Deduplicator.cs b/FileDeduplicator/Deduplicator.cs
index e43b444..f7b8591 100644
--- a/FileDeduplicator/Deduplicator.cs
+++ b/FileDeduplicator/Deduplicator.cs
@@ -28,10 +28,71 @@ internal static Dictionary> GroupByHash(Dictionar
return groups;
}
- internal static IReadOnlyList FindDuplicates(Dictionary> hashGroups) =>
- [.. hashGroups
- .Where(kvp => kvp.Value.Count > 1)
- .Select(kvp => new DuplicateGroup(kvp.Key, kvp.Value))];
+ ///
+ /// Builds the duplicate groups from the hash groups, sizing each from the copies still on disk.
+ ///
+ ///
+ /// The hash pass can take minutes on a large tree, and the directories this is pointed at are
+ /// often live, so a file that hashed may be gone by now. A vanished copy is dropped rather than
+ /// allowed to abort the run, and a group it leaves with a single copy is no longer a group.
+ ///
+ /// The files that hashed, grouped by hash.
+ /// Every group that still has at least two copies.
+ internal static IReadOnlyList FindDuplicates(Dictionary> hashGroups)
+ {
+ List duplicates = [];
+
+ foreach (KeyValuePair> kvp in hashGroups.Where(kvp => kvp.Value.Count > 1))
+ {
+ // Each copy is sized exactly once, so the filter and the size it reports cannot disagree
+ // about a file that disappears between two reads.
+ (AbsoluteFilePath File, long Size)[] present = [.. kvp.Value
+ .Select(file => (File: file, Size: TryGetSize(file, out long size) ? size : (long?)null))
+ .Where(copy => copy.Size.HasValue)
+ .Select(copy => (copy.File, copy.Size!.Value))];
+
+ if (present.Length > 1)
+ {
+ duplicates.Add(new DuplicateGroup(kvp.Key, [.. present.Select(copy => copy.File)], present[0].Size));
+ }
+ }
+
+ return duplicates;
+ }
+
+ ///
+ /// Sums the sizes of the given files, counting nothing for one that is no longer there.
+ ///
+ /// The files to size.
+ /// The total size in bytes of the files that could still be sized.
+ internal static long TotalSize(IEnumerable files) =>
+ files.Sum(file => TryGetSize(file, out long size) ? size : 0);
+
+ ///
+ /// Reads a file's size, tolerating a file that has disappeared or can no longer be reached.
+ ///
+ /// The file to size.
+ /// The size in bytes, when it could be read.
+ /// if the file is still there and its size was read.
+ private static bool TryGetSize(AbsoluteFilePath file, out long size)
+ {
+ try
+ {
+ size = new FileInfo(file.WeakString).Length;
+ return true;
+ }
+ catch (IOException)
+ {
+ // FileNotFoundException and DirectoryNotFoundException both derive from this.
+ size = 0;
+ return false;
+ }
+ catch (UnauthorizedAccessException)
+ {
+ size = 0;
+ return false;
+ }
+ }
internal static AbsoluteFilePath SelectFileToKeep(List duplicates) =>
duplicates.OrderBy(f => f.FileName.WeakString.Length).ThenBy(f => f.WeakString, StringComparer.Ordinal).First();
@@ -155,11 +216,11 @@ private static void Skip(AbsoluteFilePath file, string? reason, List files)
+internal sealed class DuplicateGroup(string hash, List files, long fileSize)
{
internal string Hash { get; } = hash;
internal List Files { get; } = files;
- internal long FileSize { get; } = new FileInfo(files[0].WeakString).Length;
+ internal long FileSize { get; } = fileSize;
}
internal sealed class DeduplicationResult(int deletedCount, long bytesReclaimed, List errors, List skippedFiles)
diff --git a/FileDeduplicator/Verbs/Stats.cs b/FileDeduplicator/Verbs/Stats.cs
index cef01c6..04fe298 100644
--- a/FileDeduplicator/Verbs/Stats.cs
+++ b/FileDeduplicator/Verbs/Stats.cs
@@ -58,7 +58,7 @@ internal override void Run(Stats options)
// Every count below is taken over the files that hashed. HashFiles drops a file it cannot
// read, so counting against the scanned list would report each one as a duplicate of nothing,
// and sizing it would throw if it had vanished since the scan.
- long totalSize = fileHashes.Keys.Sum(f => new FileInfo(f.WeakString).Length);
+ long totalSize = Deduplicator.TotalSize(fileHashes.Keys);
int uniqueFiles = hashGroups.Count;
int duplicateFiles = fileHashes.Count - uniqueFiles;
int unreadableFiles = files.Count - fileHashes.Count;