From 6c5a058809c8f5a783e19fb01020c7a6008d7095 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 16:31:24 +0000 Subject: [PATCH 1/2] fix: tolerate a file that vanishes during the hash pass [patch] After hashing, FindDuplicates sized each group with FileInfo.Length on its first file, and Stats summed FileInfo.Length over every hashed file. A file removed or renamed during a multi-minute hash pass made either throw an uncaught FileNotFoundException, losing the whole DryRun, Deduplicate, Scan or Stats run -- often over an unrelated file. FindDuplicates now sizes each copy tolerantly, drops a copy that has vanished, and drops a group left with fewer than two copies. DuplicateGroup takes its size from that pass instead of reading the disk itself. Stats sums sizes through the same tolerant lookup, counting nothing for a vanished file. Fixes ktsu-dev/FileDeduplicator#141 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_015LsGhhi4Vq9pNq2fr5T75T --- FileDeduplicator.Test/DeduplicatorTests.cs | 69 +++++++++++++++++++ FileDeduplicator/Deduplicator.cs | 78 ++++++++++++++++++++-- FileDeduplicator/Verbs/Stats.cs | 2 +- 3 files changed, 142 insertions(+), 7 deletions(-) diff --git a/FileDeduplicator.Test/DeduplicatorTests.cs b/FileDeduplicator.Test/DeduplicatorTests.cs index f792498..f1a6318 100644 --- a/FileDeduplicator.Test/DeduplicatorTests.cs +++ b/FileDeduplicator.Test/DeduplicatorTests.cs @@ -432,4 +432,73 @@ public void DeletingAnEmptyGroupListIsANoOp() Assert.IsEmpty(result.Errors); Assert.IsEmpty(result.SkippedFiles); } + + /// + /// A copy removed between the hash pass and grouping must not abort the run: the grouping + /// has to go on without it and still report the copies that remain. + /// + [TestMethod] + public void ACopyThatVanishedAfterHashingIsDroppedFromItsGroup() + { + // Arrange + using TempTree tree = new(); + AbsoluteFilePath first = tree.Write("a.txt", "shared"); + AbsoluteFilePath second = tree.Write("bb.txt", "shared"); + AbsoluteFilePath vanishing = tree.Write("ccc.txt", "shared"); + Dictionary hashes = FileHasher.HashFiles([first, second, vanishing]); + + // Arrange -- something else removes it while the rest of a long hash pass runs + File.Delete(vanishing.WeakString); + + // Act + IReadOnlyList duplicates = Duplicates(hashes); + + // Assert + Assert.ContainsSingle(duplicates); + CollectionAssert.AreEquivalent(new[] { first, second }, duplicates[0].Files); + Assert.AreEqual("shared".Length, duplicates[0].FileSize); + } + + /// + /// Whichever copy vanishes, including the one the group would have been sized from, a group + /// left with a single copy is no longer a duplicate group. + /// + /// Which of the two copies disappears. + [TestMethod] + [DataRow(0)] + [DataRow(1)] + public void AGroupLeftWithOneCopyIsDropped(int vanishingIndex) + { + // Arrange + using TempTree tree = new(); + AbsoluteFilePath[] copies = [tree.Write("a.txt", "shared"), tree.Write("bb.txt", "shared")]; + Dictionary hashes = FileHasher.HashFiles(copies); + File.Delete(copies[vanishingIndex].WeakString); + + // Act + IReadOnlyList duplicates = Duplicates(hashes); + + // Assert + Assert.IsEmpty(duplicates); + } + + /// + /// Stats sizes every file that hashed; one that has since disappeared contributes nothing + /// rather than aborting the report. + /// + [TestMethod] + public void TotalSizeSkipsAFileThatVanishedAfterHashing() + { + // Arrange + using TempTree tree = new(); + AbsoluteFilePath kept = tree.Write("a.txt", "four"); + AbsoluteFilePath vanishing = tree.Write("c.txt", "unrelated"); + File.Delete(vanishing.WeakString); + + // Act + long total = Deduplicator.TotalSize([kept, vanishing]); + + // Assert + Assert.AreEqual("four".Length, total); + } } diff --git a/FileDeduplicator/Deduplicator.cs b/FileDeduplicator/Deduplicator.cs index e43b444..47e1c84 100644 --- a/FileDeduplicator/Deduplicator.cs +++ b/FileDeduplicator/Deduplicator.cs @@ -28,10 +28,76 @@ internal static Dictionary> GroupByHash(Dictionar return groups; } - internal static IReadOnlyList FindDuplicates(Dictionary> hashGroups) => - [.. hashGroups - .Where(kvp => kvp.Value.Count > 1) - .Select(kvp => new DuplicateGroup(kvp.Key, kvp.Value))]; + /// + /// Builds the duplicate groups from the hash groups, sizing each from the copies still on disk. + /// + /// + /// The hash pass can take minutes on a large tree, and the directories this is pointed at are + /// often live, so a file that hashed may be gone by now. A vanished copy is dropped rather than + /// allowed to abort the run, and a group it leaves with a single copy is no longer a group. + /// + /// The files that hashed, grouped by hash. + /// Every group that still has at least two copies. + internal static IReadOnlyList FindDuplicates(Dictionary> hashGroups) + { + List duplicates = []; + + foreach (KeyValuePair> kvp in hashGroups.Where(kvp => kvp.Value.Count > 1)) + { + List present = []; + long fileSize = 0; + + foreach (AbsoluteFilePath file in kvp.Value) + { + if (TryGetSize(file, out long size)) + { + present.Add(file); + fileSize = size; + } + } + + if (present.Count > 1) + { + duplicates.Add(new DuplicateGroup(kvp.Key, present, fileSize)); + } + } + + return duplicates; + } + + /// + /// Sums the sizes of the given files, counting nothing for one that is no longer there. + /// + /// The files to size. + /// The total size in bytes of the files that could still be sized. + internal static long TotalSize(IEnumerable files) => + files.Sum(file => TryGetSize(file, out long size) ? size : 0); + + /// + /// Reads a file's size, tolerating a file that has disappeared or can no longer be reached. + /// + /// The file to size. + /// The size in bytes, when it could be read. + /// if the file is still there and its size was read. + private static bool TryGetSize(AbsoluteFilePath file, out long size) + { + try + { + size = new FileInfo(file.WeakString).Length; + return true; + } + catch (IOException) + { + // FileNotFoundException and DirectoryNotFoundException both derive from this. + size = 0; + return false; + } + catch (UnauthorizedAccessException) + { + size = 0; + return false; + } + } internal static AbsoluteFilePath SelectFileToKeep(List duplicates) => duplicates.OrderBy(f => f.FileName.WeakString.Length).ThenBy(f => f.WeakString, StringComparer.Ordinal).First(); @@ -155,11 +221,11 @@ private static void Skip(AbsoluteFilePath file, string? reason, List files) +internal sealed class DuplicateGroup(string hash, List files, long fileSize) { internal string Hash { get; } = hash; internal List Files { get; } = files; - internal long FileSize { get; } = new FileInfo(files[0].WeakString).Length; + internal long FileSize { get; } = fileSize; } internal sealed class DeduplicationResult(int deletedCount, long bytesReclaimed, List errors, List skippedFiles) diff --git a/FileDeduplicator/Verbs/Stats.cs b/FileDeduplicator/Verbs/Stats.cs index cef01c6..04fe298 100644 --- a/FileDeduplicator/Verbs/Stats.cs +++ b/FileDeduplicator/Verbs/Stats.cs @@ -58,7 +58,7 @@ internal override void Run(Stats options) // Every count below is taken over the files that hashed. HashFiles drops a file it cannot // read, so counting against the scanned list would report each one as a duplicate of nothing, // and sizing it would throw if it had vanished since the scan. - long totalSize = fileHashes.Keys.Sum(f => new FileInfo(f.WeakString).Length); + long totalSize = Deduplicator.TotalSize(fileHashes.Keys); int uniqueFiles = hashGroups.Count; int duplicateFiles = fileHashes.Count - uniqueFiles; int unreadableFiles = files.Count - fileHashes.Count; From 4d92de87a9fcd691562223434524fd43d24c1292 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 16:33:48 +0000 Subject: [PATCH 2/2] Filter vanished copies with an explicit Where, sizing each copy once Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_015LsGhhi4Vq9pNq2fr5T75T --- FileDeduplicator/Deduplicator.cs | 23 +++++++++-------------- 1 file changed, 9 insertions(+), 14 deletions(-) diff --git a/FileDeduplicator/Deduplicator.cs b/FileDeduplicator/Deduplicator.cs index 47e1c84..f7b8591 100644 --- a/FileDeduplicator/Deduplicator.cs +++ b/FileDeduplicator/Deduplicator.cs @@ -44,21 +44,16 @@ internal static IReadOnlyList FindDuplicates(Dictionary> kvp in hashGroups.Where(kvp => kvp.Value.Count > 1)) { - List present = []; - long fileSize = 0; - - foreach (AbsoluteFilePath file in kvp.Value) - { - if (TryGetSize(file, out long size)) - { - present.Add(file); - fileSize = size; - } - } - - if (present.Count > 1) + // Each copy is sized exactly once, so the filter and the size it reports cannot disagree + // about a file that disappears between two reads. + (AbsoluteFilePath File, long Size)[] present = [.. kvp.Value + .Select(file => (File: file, Size: TryGetSize(file, out long size) ? size : (long?)null)) + .Where(copy => copy.Size.HasValue) + .Select(copy => (copy.File, copy.Size!.Value))]; + + if (present.Length > 1) { - duplicates.Add(new DuplicateGroup(kvp.Key, present, fileSize)); + duplicates.Add(new DuplicateGroup(kvp.Key, [.. present.Select(copy => copy.File)], present[0].Size)); } }