diff --git a/FileDeduplicator.Test/DeduplicatorTests.cs b/FileDeduplicator.Test/DeduplicatorTests.cs index f792498..f1a6318 100644 --- a/FileDeduplicator.Test/DeduplicatorTests.cs +++ b/FileDeduplicator.Test/DeduplicatorTests.cs @@ -432,4 +432,73 @@ public void DeletingAnEmptyGroupListIsANoOp() Assert.IsEmpty(result.Errors); Assert.IsEmpty(result.SkippedFiles); } + + /// + /// A copy removed between the hash pass and grouping must not abort the run: the grouping + /// has to go on without it and still report the copies that remain. + /// + [TestMethod] + public void ACopyThatVanishedAfterHashingIsDroppedFromItsGroup() + { + // Arrange + using TempTree tree = new(); + AbsoluteFilePath first = tree.Write("a.txt", "shared"); + AbsoluteFilePath second = tree.Write("bb.txt", "shared"); + AbsoluteFilePath vanishing = tree.Write("ccc.txt", "shared"); + Dictionary hashes = FileHasher.HashFiles([first, second, vanishing]); + + // Arrange -- something else removes it while the rest of a long hash pass runs + File.Delete(vanishing.WeakString); + + // Act + IReadOnlyList duplicates = Duplicates(hashes); + + // Assert + Assert.ContainsSingle(duplicates); + CollectionAssert.AreEquivalent(new[] { first, second }, duplicates[0].Files); + Assert.AreEqual("shared".Length, duplicates[0].FileSize); + } + + /// + /// Whichever copy vanishes, including the one the group would have been sized from, a group + /// left with a single copy is no longer a duplicate group. + /// + /// Which of the two copies disappears. + [TestMethod] + [DataRow(0)] + [DataRow(1)] + public void AGroupLeftWithOneCopyIsDropped(int vanishingIndex) + { + // Arrange + using TempTree tree = new(); + AbsoluteFilePath[] copies = [tree.Write("a.txt", "shared"), tree.Write("bb.txt", "shared")]; + Dictionary hashes = FileHasher.HashFiles(copies); + File.Delete(copies[vanishingIndex].WeakString); + + // Act + IReadOnlyList duplicates = Duplicates(hashes); + + // Assert + Assert.IsEmpty(duplicates); + } + + /// + /// Stats sizes every file that hashed; one that has since disappeared contributes nothing + /// rather than aborting the report. + /// + [TestMethod] + public void TotalSizeSkipsAFileThatVanishedAfterHashing() + { + // Arrange + using TempTree tree = new(); + AbsoluteFilePath kept = tree.Write("a.txt", "four"); + AbsoluteFilePath vanishing = tree.Write("c.txt", "unrelated"); + File.Delete(vanishing.WeakString); + + // Act + long total = Deduplicator.TotalSize([kept, vanishing]); + + // Assert + Assert.AreEqual("four".Length, total); + } } diff --git a/FileDeduplicator/Deduplicator.cs b/FileDeduplicator/Deduplicator.cs index e43b444..f7b8591 100644 --- a/FileDeduplicator/Deduplicator.cs +++ b/FileDeduplicator/Deduplicator.cs @@ -28,10 +28,71 @@ internal static Dictionary> GroupByHash(Dictionar return groups; } - internal static IReadOnlyList FindDuplicates(Dictionary> hashGroups) => - [.. hashGroups - .Where(kvp => kvp.Value.Count > 1) - .Select(kvp => new DuplicateGroup(kvp.Key, kvp.Value))]; + /// + /// Builds the duplicate groups from the hash groups, sizing each from the copies still on disk. + /// + /// + /// The hash pass can take minutes on a large tree, and the directories this is pointed at are + /// often live, so a file that hashed may be gone by now. A vanished copy is dropped rather than + /// allowed to abort the run, and a group it leaves with a single copy is no longer a group. + /// + /// The files that hashed, grouped by hash. + /// Every group that still has at least two copies. + internal static IReadOnlyList FindDuplicates(Dictionary> hashGroups) + { + List duplicates = []; + + foreach (KeyValuePair> kvp in hashGroups.Where(kvp => kvp.Value.Count > 1)) + { + // Each copy is sized exactly once, so the filter and the size it reports cannot disagree + // about a file that disappears between two reads. + (AbsoluteFilePath File, long Size)[] present = [.. kvp.Value + .Select(file => (File: file, Size: TryGetSize(file, out long size) ? size : (long?)null)) + .Where(copy => copy.Size.HasValue) + .Select(copy => (copy.File, copy.Size!.Value))]; + + if (present.Length > 1) + { + duplicates.Add(new DuplicateGroup(kvp.Key, [.. present.Select(copy => copy.File)], present[0].Size)); + } + } + + return duplicates; + } + + /// + /// Sums the sizes of the given files, counting nothing for one that is no longer there. + /// + /// The files to size. + /// The total size in bytes of the files that could still be sized. + internal static long TotalSize(IEnumerable files) => + files.Sum(file => TryGetSize(file, out long size) ? size : 0); + + /// + /// Reads a file's size, tolerating a file that has disappeared or can no longer be reached. + /// + /// The file to size. + /// The size in bytes, when it could be read. + /// if the file is still there and its size was read. + private static bool TryGetSize(AbsoluteFilePath file, out long size) + { + try + { + size = new FileInfo(file.WeakString).Length; + return true; + } + catch (IOException) + { + // FileNotFoundException and DirectoryNotFoundException both derive from this. + size = 0; + return false; + } + catch (UnauthorizedAccessException) + { + size = 0; + return false; + } + } internal static AbsoluteFilePath SelectFileToKeep(List duplicates) => duplicates.OrderBy(f => f.FileName.WeakString.Length).ThenBy(f => f.WeakString, StringComparer.Ordinal).First(); @@ -155,11 +216,11 @@ private static void Skip(AbsoluteFilePath file, string? reason, List files) +internal sealed class DuplicateGroup(string hash, List files, long fileSize) { internal string Hash { get; } = hash; internal List Files { get; } = files; - internal long FileSize { get; } = new FileInfo(files[0].WeakString).Length; + internal long FileSize { get; } = fileSize; } internal sealed class DeduplicationResult(int deletedCount, long bytesReclaimed, List errors, List skippedFiles) diff --git a/FileDeduplicator/Verbs/Stats.cs b/FileDeduplicator/Verbs/Stats.cs index cef01c6..04fe298 100644 --- a/FileDeduplicator/Verbs/Stats.cs +++ b/FileDeduplicator/Verbs/Stats.cs @@ -58,7 +58,7 @@ internal override void Run(Stats options) // Every count below is taken over the files that hashed. HashFiles drops a file it cannot // read, so counting against the scanned list would report each one as a duplicate of nothing, // and sizing it would throw if it had vanished since the scan. - long totalSize = fileHashes.Keys.Sum(f => new FileInfo(f.WeakString).Length); + long totalSize = Deduplicator.TotalSize(fileHashes.Keys); int uniqueFiles = hashGroups.Count; int duplicateFiles = fileHashes.Count - uniqueFiles; int unreadableFiles = files.Count - fileHashes.Count;